fieldtrial 0.1.0.dev0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (181) hide show
  1. fieldtrial-0.2.0/.gitignore +39 -0
  2. fieldtrial-0.2.0/CHANGELOG.md +181 -0
  3. fieldtrial-0.2.0/PKG-INFO +204 -0
  4. fieldtrial-0.2.0/README.md +137 -0
  5. fieldtrial-0.2.0/pyproject.toml +368 -0
  6. fieldtrial-0.2.0/src/fieldtrial/__init__.py +13 -0
  7. fieldtrial-0.2.0/src/fieldtrial/analysis/__init__.py +1 -0
  8. fieldtrial-0.2.0/src/fieldtrial/analysis/common.py +20 -0
  9. fieldtrial-0.2.0/src/fieldtrial/analysis/crossover.py +83 -0
  10. fieldtrial-0.2.0/src/fieldtrial/analysis/engine.py +1222 -0
  11. fieldtrial-0.2.0/src/fieldtrial/analysis/ladder.py +103 -0
  12. fieldtrial-0.2.0/src/fieldtrial/analysis/records.py +57 -0
  13. fieldtrial-0.2.0/src/fieldtrial/analysis/results.py +409 -0
  14. fieldtrial-0.2.0/src/fieldtrial/analysis/sequential.py +281 -0
  15. fieldtrial-0.2.0/src/fieldtrial/analysis/wording.py +400 -0
  16. fieldtrial-0.2.0/src/fieldtrial/api/__init__.py +1 -0
  17. fieldtrial-0.2.0/src/fieldtrial/api/schemas.py +213 -0
  18. fieldtrial-0.2.0/src/fieldtrial/api/v1.py +494 -0
  19. fieldtrial-0.2.0/src/fieldtrial/capture/__init__.py +1 -0
  20. fieldtrial-0.2.0/src/fieldtrial/capture/drift.py +193 -0
  21. fieldtrial-0.2.0/src/fieldtrial/capture/images.py +32 -0
  22. fieldtrial-0.2.0/src/fieldtrial/capture/recorder.py +208 -0
  23. fieldtrial-0.2.0/src/fieldtrial/cli/__init__.py +1 -0
  24. fieldtrial-0.2.0/src/fieldtrial/cli/calc.py +299 -0
  25. fieldtrial-0.2.0/src/fieldtrial/cli/main.py +42 -0
  26. fieldtrial-0.2.0/src/fieldtrial/cli/serve.py +112 -0
  27. fieldtrial-0.2.0/src/fieldtrial/cli/study.py +536 -0
  28. fieldtrial-0.2.0/src/fieldtrial/client.py +283 -0
  29. fieldtrial-0.2.0/src/fieldtrial/design/__init__.py +32 -0
  30. fieldtrial-0.2.0/src/fieldtrial/design/blinding.py +26 -0
  31. fieldtrial-0.2.0/src/fieldtrial/design/capture_config.py +38 -0
  32. fieldtrial-0.2.0/src/fieldtrial/design/hashing.py +47 -0
  33. fieldtrial-0.2.0/src/fieldtrial/design/loader.py +135 -0
  34. fieldtrial-0.2.0/src/fieldtrial/design/models.py +396 -0
  35. fieldtrial-0.2.0/src/fieldtrial/design/runner_config.py +171 -0
  36. fieldtrial-0.2.0/src/fieldtrial/design/schedule.py +170 -0
  37. fieldtrial-0.2.0/src/fieldtrial/io/__init__.py +1 -0
  38. fieldtrial-0.2.0/src/fieldtrial/io/csv.py +219 -0
  39. fieldtrial-0.2.0/src/fieldtrial/io/jsonl.py +17 -0
  40. fieldtrial-0.2.0/src/fieldtrial/io/lerobot.py +133 -0
  41. fieldtrial-0.2.0/src/fieldtrial/py.typed +0 -0
  42. fieldtrial-0.2.0/src/fieldtrial/report/__init__.py +6 -0
  43. fieldtrial-0.2.0/src/fieldtrial/report/charts.py +462 -0
  44. fieldtrial-0.2.0/src/fieldtrial/report/html.py +79 -0
  45. fieldtrial-0.2.0/src/fieldtrial/report/markdown.py +493 -0
  46. fieldtrial-0.2.0/src/fieldtrial/report/templates/report.css +25 -0
  47. fieldtrial-0.2.0/src/fieldtrial/report/templates/report.html +280 -0
  48. fieldtrial-0.2.0/src/fieldtrial/runners/__init__.py +1 -0
  49. fieldtrial-0.2.0/src/fieldtrial/runners/base.py +98 -0
  50. fieldtrial-0.2.0/src/fieldtrial/runners/command.py +311 -0
  51. fieldtrial-0.2.0/src/fieldtrial/runners/manual.py +55 -0
  52. fieldtrial-0.2.0/src/fieldtrial/runners/openpi_router.py +315 -0
  53. fieldtrial-0.2.0/src/fieldtrial/runners/sim.py +133 -0
  54. fieldtrial-0.2.0/src/fieldtrial/services/__init__.py +14 -0
  55. fieldtrial-0.2.0/src/fieldtrial/services/_context.py +61 -0
  56. fieldtrial-0.2.0/src/fieldtrial/services/analysis.py +33 -0
  57. fieldtrial-0.2.0/src/fieldtrial/services/dataset.py +152 -0
  58. fieldtrial-0.2.0/src/fieldtrial/services/events.py +110 -0
  59. fieldtrial-0.2.0/src/fieldtrial/services/interim.py +133 -0
  60. fieldtrial-0.2.0/src/fieldtrial/services/registry.py +95 -0
  61. fieldtrial-0.2.0/src/fieldtrial/services/rig.py +170 -0
  62. fieldtrial-0.2.0/src/fieldtrial/services/runner_check.py +96 -0
  63. fieldtrial-0.2.0/src/fieldtrial/services/session.py +206 -0
  64. fieldtrial-0.2.0/src/fieldtrial/services/simulate.py +197 -0
  65. fieldtrial-0.2.0/src/fieldtrial/services/study.py +508 -0
  66. fieldtrial-0.2.0/src/fieldtrial/services/transfer.py +136 -0
  67. fieldtrial-0.2.0/src/fieldtrial/services/trial.py +940 -0
  68. fieldtrial-0.2.0/src/fieldtrial/stats/__init__.py +109 -0
  69. fieldtrial-0.2.0/src/fieldtrial/stats/_rng.py +72 -0
  70. fieldtrial-0.2.0/src/fieldtrial/stats/_types.py +237 -0
  71. fieldtrial-0.2.0/src/fieldtrial/stats/_validation.py +48 -0
  72. fieldtrial-0.2.0/src/fieldtrial/stats/bayes.py +82 -0
  73. fieldtrial-0.2.0/src/fieldtrial/stats/compare.py +206 -0
  74. fieldtrial-0.2.0/src/fieldtrial/stats/crossover.py +247 -0
  75. fieldtrial-0.2.0/src/fieldtrial/stats/drift.py +80 -0
  76. fieldtrial-0.2.0/src/fieldtrial/stats/ladder.py +357 -0
  77. fieldtrial-0.2.0/src/fieldtrial/stats/multiplicity.py +96 -0
  78. fieldtrial-0.2.0/src/fieldtrial/stats/ordinal.py +139 -0
  79. fieldtrial-0.2.0/src/fieldtrial/stats/paired.py +244 -0
  80. fieldtrial-0.2.0/src/fieldtrial/stats/power.py +501 -0
  81. fieldtrial-0.2.0/src/fieldtrial/stats/proportions.py +160 -0
  82. fieldtrial-0.2.0/src/fieldtrial/stats/sequential.py +475 -0
  83. fieldtrial-0.2.0/src/fieldtrial/stats/stratified.py +127 -0
  84. fieldtrial-0.2.0/src/fieldtrial/stats/timing.py +103 -0
  85. fieldtrial-0.2.0/src/fieldtrial/store/__init__.py +1 -0
  86. fieldtrial-0.2.0/src/fieldtrial/store/db.py +46 -0
  87. fieldtrial-0.2.0/src/fieldtrial/store/ids.py +17 -0
  88. fieldtrial-0.2.0/src/fieldtrial/store/migrations/__init__.py +0 -0
  89. fieldtrial-0.2.0/src/fieldtrial/store/migrations/env.py +14 -0
  90. fieldtrial-0.2.0/src/fieldtrial/store/migrations/script.py.mako +21 -0
  91. fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/0001_baseline.py +182 -0
  92. fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/0002_idempotency.py +24 -0
  93. fieldtrial-0.2.0/src/fieldtrial/store/migrations/versions/__init__.py +0 -0
  94. fieldtrial-0.2.0/src/fieldtrial/store/models.py +188 -0
  95. fieldtrial-0.2.0/src/fieldtrial/templates/__init__.py +15 -0
  96. fieldtrial-0.2.0/src/fieldtrial/templates/basic.yaml +58 -0
  97. fieldtrial-0.2.0/src/fieldtrial/templates/checkpoint-ladder.yaml +45 -0
  98. fieldtrial-0.2.0/src/fieldtrial/templates/crossover-rounds.yaml +43 -0
  99. fieldtrial-0.2.0/src/fieldtrial/templates/demo.yaml +59 -0
  100. fieldtrial-0.2.0/src/fieldtrial/templates/serving-sweep.yaml +41 -0
  101. fieldtrial-0.2.0/src/fieldtrial/web/__init__.py +1 -0
  102. fieldtrial-0.2.0/src/fieldtrial/web/app.py +98 -0
  103. fieldtrial-0.2.0/src/fieldtrial/web/console.py +536 -0
  104. fieldtrial-0.2.0/src/fieldtrial/web/runners.py +340 -0
  105. fieldtrial-0.2.0/src/fieldtrial/web/security.py +169 -0
  106. fieldtrial-0.2.0/src/fieldtrial/web/static/app.css +174 -0
  107. fieldtrial-0.2.0/src/fieldtrial/web/static/app.js +216 -0
  108. fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/README.md +11 -0
  109. fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/htmx-2.0.11.min.js +1 -0
  110. fieldtrial-0.2.0/src/fieldtrial/web/static/vendor/htmx-LICENSE.txt +13 -0
  111. fieldtrial-0.2.0/src/fieldtrial/web/templates/_edit_row.html +38 -0
  112. fieldtrial-0.2.0/src/fieldtrial/web/templates/_flash.html +1 -0
  113. fieldtrial-0.2.0/src/fieldtrial/web/templates/_panel.html +125 -0
  114. fieldtrial-0.2.0/src/fieldtrial/web/templates/_slot_header.html +12 -0
  115. fieldtrial-0.2.0/src/fieldtrial/web/templates/_trial_header.html +2 -0
  116. fieldtrial-0.2.0/src/fieldtrial/web/templates/_trial_row.html +12 -0
  117. fieldtrial-0.2.0/src/fieldtrial/web/templates/base.html +24 -0
  118. fieldtrial-0.2.0/src/fieldtrial/web/templates/console.html +27 -0
  119. fieldtrial-0.2.0/src/fieldtrial/web/templates/draft.html +13 -0
  120. fieldtrial-0.2.0/src/fieldtrial/web/templates/error.html +6 -0
  121. fieldtrial-0.2.0/src/fieldtrial/web/templates/history.html +20 -0
  122. fieldtrial-0.2.0/src/fieldtrial/web/templates/report.html +54 -0
  123. fieldtrial-0.2.0/src/fieldtrial/web/templates/studies.html +19 -0
  124. fieldtrial-0.2.0/src/fieldtrial/web/templates/study.html +85 -0
  125. fieldtrial-0.2.0/tests/analysis/test_engine.py +361 -0
  126. fieldtrial-0.2.0/tests/analysis/test_report.py +72 -0
  127. fieldtrial-0.2.0/tests/analysis/test_schema.py +19 -0
  128. fieldtrial-0.2.0/tests/analysis/test_wording.py +210 -0
  129. fieldtrial-0.2.0/tests/capture/test_drift.py +118 -0
  130. fieldtrial-0.2.0/tests/capture/test_recorder.py +133 -0
  131. fieldtrial-0.2.0/tests/conftest.py +50 -0
  132. fieldtrial-0.2.0/tests/design/snapshots/schedules.json +3580 -0
  133. fieldtrial-0.2.0/tests/design/test_models.py +280 -0
  134. fieldtrial-0.2.0/tests/design/test_schedule.py +232 -0
  135. fieldtrial-0.2.0/tests/fixtures/golden/study.yaml +30 -0
  136. fieldtrial-0.2.0/tests/fixtures/golden/trials.csv +201 -0
  137. fieldtrial-0.2.0/tests/io/test_lerobot_links.py +161 -0
  138. fieldtrial-0.2.0/tests/report/test_html.py +166 -0
  139. fieldtrial-0.2.0/tests/runners/test_command.py +274 -0
  140. fieldtrial-0.2.0/tests/runners/test_openpi_router.py +200 -0
  141. fieldtrial-0.2.0/tests/services/conftest.py +31 -0
  142. fieldtrial-0.2.0/tests/services/test_console_services.py +272 -0
  143. fieldtrial-0.2.0/tests/services/test_registry.py +41 -0
  144. fieldtrial-0.2.0/tests/services/test_rig.py +109 -0
  145. fieldtrial-0.2.0/tests/services/test_simulate.py +153 -0
  146. fieldtrial-0.2.0/tests/services/test_store.py +80 -0
  147. fieldtrial-0.2.0/tests/services/test_study_services.py +164 -0
  148. fieldtrial-0.2.0/tests/services/test_transfer.py +119 -0
  149. fieldtrial-0.2.0/tests/services/test_trial_services.py +194 -0
  150. fieldtrial-0.2.0/tests/services/test_v02_designs.py +238 -0
  151. fieldtrial-0.2.0/tests/stats/test_bayes.py +71 -0
  152. fieldtrial-0.2.0/tests/stats/test_compare.py +159 -0
  153. fieldtrial-0.2.0/tests/stats/test_crossover.py +180 -0
  154. fieldtrial-0.2.0/tests/stats/test_ladder.py +205 -0
  155. fieldtrial-0.2.0/tests/stats/test_m2_stats.py +214 -0
  156. fieldtrial-0.2.0/tests/stats/test_multiplicity.py +91 -0
  157. fieldtrial-0.2.0/tests/stats/test_paired.py +191 -0
  158. fieldtrial-0.2.0/tests/stats/test_power.py +252 -0
  159. fieldtrial-0.2.0/tests/stats/test_proportions.py +166 -0
  160. fieldtrial-0.2.0/tests/stats/test_sequential.py +218 -0
  161. fieldtrial-0.2.0/tests/stats/test_slow_properties.py +73 -0
  162. fieldtrial-0.2.0/tests/test_architecture.py +110 -0
  163. fieldtrial-0.2.0/tests/test_calc_cli.py +150 -0
  164. fieldtrial-0.2.0/tests/test_cli.py +48 -0
  165. fieldtrial-0.2.0/tests/test_examples.py +66 -0
  166. fieldtrial-0.2.0/tests/test_network_guard.py +27 -0
  167. fieldtrial-0.2.0/tests/test_release_tools.py +55 -0
  168. fieldtrial-0.2.0/tests/test_study_cli.py +146 -0
  169. fieldtrial-0.2.0/tests/web/conftest.py +33 -0
  170. fieldtrial-0.2.0/tests/web/test_api.py +272 -0
  171. fieldtrial-0.2.0/tests/web/test_capture_web.py +129 -0
  172. fieldtrial-0.2.0/tests/web/test_client.py +85 -0
  173. fieldtrial-0.2.0/tests/web/test_console.py +450 -0
  174. fieldtrial-0.2.0/tests/web/test_runners_web.py +239 -0
  175. fieldtrial-0.2.0/tests/web/test_serve.py +117 -0
  176. fieldtrial-0.2.0/tests/web/test_v02_web.py +109 -0
  177. fieldtrial-0.1.0.dev0/PKG-INFO +0 -22
  178. fieldtrial-0.1.0.dev0/README.md +0 -9
  179. fieldtrial-0.1.0.dev0/pyproject.toml +0 -27
  180. fieldtrial-0.1.0.dev0/src/fieldtrial/__init__.py +0 -6
  181. {fieldtrial-0.1.0.dev0 → fieldtrial-0.2.0}/LICENSE +0 -0
@@ -0,0 +1,39 @@
1
+ # Secrets: tokens and credentials stay local
2
+ .env
3
+ .env.*
4
+ !.env.example
5
+ .pypirc
6
+
7
+ # Python bytecode
8
+ __pycache__/
9
+ *.py[cod]
10
+
11
+ # Virtual environments (uv)
12
+ .venv/
13
+ venv/
14
+
15
+ # Build artifacts (uv build / hatchling)
16
+ build/
17
+ dist/
18
+ *.egg-info/
19
+ *.whl
20
+
21
+ # Test, lint and type-check caches
22
+ .pytest_cache/
23
+ .hypothesis/
24
+ .mypy_cache/
25
+ .ruff_cache/
26
+ .import_linter_cache/
27
+ .coverage
28
+ .coverage.*
29
+ htmlcov/
30
+ coverage.xml
31
+
32
+ # Docs build (mkdocs)
33
+ /site/
34
+
35
+ # Editors and OS clutter
36
+ .idea/
37
+ .vscode/
38
+ .DS_Store
39
+ Thumbs.db
@@ -0,0 +1,181 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this
6
+ project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). Before 1.0,
7
+ minor releases may contain breaking changes.
8
+
9
+ ## [Unreleased]
10
+
11
+ ## [0.2.0] - 2026-10-02
12
+
13
+ Second release: crossover rounds, checkpoint ladders and group-sequential stopping; real
14
+ blinding with switching runners; evaluation-camera capture, rig drift checks and links to
15
+ LeRobot datasets. Studies, design hashes and results files from 0.1 keep working.
16
+
17
+ ### Added
18
+
19
+ - Crossover rounds (`design.type: crossover_rounds`) for tasks whose scene carries over
20
+ between trials. Each arm runs whole rounds in cycles of two, in randomized, balanced
21
+ order. The analysis is a period-adjusted difference with an exact randomization test,
22
+ and its interval inverts the same test. A new `crossover-rounds` template. The console
23
+ shows the round and when to reset the scene.
24
+ - Checkpoint ladders (`analysis.ladder`): Mantel's test of a linear association between
25
+ training step and success, and plateau detection by fixed-sequence non-inferiority
26
+ against the final checkpoint. Without a comparison, the association test is the
27
+ primary analysis. Reports gain a success-by-step chart and a plateau table.
28
+ - Group-sequential stopping (`analysis.stopping: {rule: group_sequential, looks: K}`):
29
+ - Lan–DeMets error-spending boundaries (O'Brien–Fleming or Pocock type).
30
+ - Interim looks via `fieldtrial interim`, the console and `POST /api/v1/studies/{study}/interim`.
31
+ A look reveals only "continue" or "stop" and cancels the remaining trials on a stop.
32
+ - A stage-wise p-value and a repeated confidence interval in the final analysis.
33
+ - The simulator runs planned looks as they come due (`--no-interim` to skip them).
34
+ - Real blinding with switching runners. The console and the REST API start and stop them
35
+ with each trial. `fieldtrial check-runners` checks the setup and names arms by blind
36
+ code only.
37
+ - `command` runner: runs `runners.command.template` for each trial with the arm's
38
+ `policy` and `serving` values. The template is split like a shell would but never
39
+ runs in one, and placeholders are checked at validation.
40
+ - The command gets its own process group and is stopped with SIGINT, then SIGTERM,
41
+ then SIGKILL. Logs go to `logs/`, and the environment carries the blind code only.
42
+ - `launch: per_arm` keeps one process per arm and sends it start and stop lines.
43
+ - With `success_exit_code`, a successful exit preselects the success stage.
44
+ - `openpi_router` runner: a websocket proxy in front of one openpi policy server per
45
+ arm.
46
+ - It refuses arms whose metadata differ, passes `Api-Key` headers through and forwards
47
+ frames unchanged to the current trial's arm.
48
+ - It records request latency per trial. Install with `fieldtrial[openpi]`.
49
+ - A runner that cannot start a trial marks it invalid with the reason, so the slot is
50
+ rescheduled.
51
+ - Reports gain a descriptive runner table per arm: requests, errors, latency and
52
+ abnormal exits.
53
+ - Rig drift checks. Compare a photo of the rig with a reference photo: shift by phase
54
+ correlation, brightness change, and structural similarity (SSIM) after alignment.
55
+ - Run them with `fieldtrial rig-check DIR PHOTO` (or `--camera`), or attach a photo
56
+ when a console session starts.
57
+ - A flagged check shows as a console warning and as a report deviation, and the report
58
+ lists every check.
59
+ - Evaluation-camera capture (`fieldtrial[capture]`: OpenCV and PyAV). With
60
+ `capture.camera` in `study.yaml`, every trial is recorded from Start to Stop and the
61
+ clip is attached to the trial. The rig is checked from the camera at session start and
62
+ every `capture.drift.every_trials` trials. A camera failure never blocks a trial.
63
+ - Links between trials and LeRobot v3.0 dataset episodes (`fieldtrial[lerobot]`:
64
+ pyarrow).
65
+ - `fieldtrial link-episodes DIR DATASET` matches trials to episodes in run order, or
66
+ from a `trial,episode_index` mapping file, and shows the plan by blind code before
67
+ writing.
68
+ - Episodes with DAgger `intervention` flags get intervention counts, and reports gain a
69
+ descriptive dataset-episodes table.
70
+ - A `capture:` section in `study.yaml`, always left out of the design hash.
71
+ - `fieldtrial.stats`:
72
+ - `spending_boundaries`, `constant_boundaries`, `crossing_probabilities`,
73
+ `sequential_test`, `repeated_interval`
74
+ - `trend_test`, `stratified_trend_test`, `plateau`, `plateau_paired`
75
+ - `crossover_test`
76
+ - `Results` gains optional `ladder`, `crossover` and `sequential` blocks, the lists
77
+ `runner`, `rig_checks` and `episodes` (empty by default), and the primary methods
78
+ `ladder`, `crossover` and `group_sequential` (schema version unchanged: additions only).
79
+
80
+ ### Changed
81
+
82
+ - `limits.reset: carry_over` is now accepted together with `design.type: crossover_rounds`.
83
+ - Design hashes of existing studies are unchanged: new settings are left out of the hash
84
+ while they are unused.
85
+ - `pillow` is declared as a dependency; it was already required by matplotlib.
86
+
87
+ ## [0.2.0rc1] - 2026-10-02
88
+
89
+ Release candidate of 0.2.0, published to PyPI for testing. Its changes are listed under
90
+ 0.2.0; the release is identical apart from this changelog.
91
+
92
+ ## [0.2.0a2] - 2026-10-02
93
+
94
+ Second v0.2 pre-release: real blinding with the command runner and the openpi router. Its
95
+ changes are listed under 0.2.0.
96
+
97
+ ## [0.2.0a1] - 2026-10-02
98
+
99
+ First v0.2 pre-release: crossover rounds, checkpoint ladders and group-sequential stopping.
100
+ Its changes are listed under 0.2.0.
101
+
102
+ ## [0.1.0] - 2026-10-02
103
+
104
+ First release. Studies: design, lock, fill and analyze end to end, run them from a
105
+ phone-friendly operator console or your own runtime, and share a self-contained report.
106
+
107
+ ### Added
108
+
109
+ - `study.yaml` schema with line-numbered validation errors, and three templates (`basic`,
110
+ `checkpoint-ladder`, `serving-sweep`).
111
+ - Randomized complete block schedules with Williams-balanced arm order and blind codes.
112
+ The schedule is identical on every platform and numpy version for a given seed.
113
+ - Locking and amendments: the design, its hash and the schedule are stored in a SQLite
114
+ database in the study folder; every change is one transaction plus one event-log row.
115
+ - Invalid trials are kept and rescheduled at the end of their block.
116
+ - A simulated runner and auto-operator, and CSV import (all or nothing, `--map`) and
117
+ CSV/JSONL export.
118
+ - The analysis engine: the primary analysis follows the locked design (McNemar with a Tango
119
+ interval, CMH, or Cochran's Q with Holm; a threshold test for single-arm studies), with
120
+ an independent Boschloo/Newcombe sensitivity analysis, stage funnels, time to success,
121
+ per-condition and per-session tables, drift and invalid-trial checks, deviations and
122
+ provenance. The `Results` model is versioned, and its JSON schema is published.
123
+ - Report wording from one tested template per situation, and a Markdown report.
124
+ - New statistics: stage distribution and funnel, Brunner–Munzel on stages, success-time
125
+ curve and bootstrap median time, CMH test, and homogeneity tests for drift.
126
+ - Commands `init`, `validate`, `plan`, `lock`, `amend`, `simulate`, `import`, `export`,
127
+ `status`, `unblind`, `analyze` and `report`.
128
+ - Documentation: "Running a study" and "Analysis and reports".
129
+ - The operator console (`fieldtrial serve`): start a session with the rig checklist, run
130
+ trials by blind code with a timer, label the furthest stage, termination and failure tags,
131
+ undo within 10 seconds, mark invalid trials, correct labels with a logged reason, a live
132
+ mirror screen, keyboard and foot-pedal keys, light and dark themes. `--lan` serves phones
133
+ on the local network behind an access token with a QR code.
134
+ - `fieldtrial demo`: a half-run simulated study in the console.
135
+ - REST API v1 with an OpenAPI document, Server-Sent Events, idempotency keys and optimistic
136
+ concurrency, and `fieldtrial.client`, a dependency-free Python client.
137
+ - Reports list label corrections made after unblinding as deviations.
138
+ - Documentation: "Operator console", "REST API and client" and a phone test checklist.
139
+ - A self-contained HTML report with seven charts (`fieldtrial report`, now the default
140
+ format, and in the console): no scripts and nothing loaded from other hosts.
141
+ - An example re-analysis of Dream Machines' published pi0.5 fine-tuning results.
142
+ - Documentation: quickstart, concepts, guides (planning, comparing checkpoints, ladders,
143
+ serving sweeps, LeRobot, openpi, custom runtimes), and screenshots.
144
+ - `CITATION.cff`.
145
+
146
+ ## [0.1.0rc1] - 2026-10-02
147
+
148
+ Release candidate of 0.1.0, published to PyPI for testing. Its changes are listed under
149
+ 0.1.0; the release is identical apart from this changelog.
150
+
151
+ ## [0.1.0a1] - 2026-10-01
152
+
153
+ First alpha: the statistics core and the calculator commands.
154
+
155
+ ### Added
156
+
157
+ - `fieldtrial.stats`, the statistics core (numpy and scipy only):
158
+ - one-arm intervals: Wilson (default), Clopper–Pearson, Jeffreys, Agresti–Coull, and an
159
+ exact binomial test against a threshold
160
+ - two independent arms: Newcombe difference CI, Boschloo exact test (primary), Fisher's
161
+ exact test and the conditional odds ratio
162
+ - paired designs: exact McNemar, Tango score CI, Cochran's Q with pairwise McNemar
163
+ - multiplicity adjustments: Holm, Bonferroni, Benjamini–Hochberg
164
+ - planning: sample size, power and minimum detectable effect (pooled-z, Fleiss
165
+ continuity-corrected, arcsine; unequal allocation), exact power for Boschloo and
166
+ McNemar, and a seeded simulation cross-check
167
+ - Bayesian summaries (descriptive only): P(p_B > p_A) and credible intervals
168
+ - Calculator commands `fieldtrial ci`, `compare`, `paired`, `power`, `mde` and `adjust`,
169
+ each with `--json` output.
170
+ - Documentation: a statistics reference page per method, a command-line page and the API
171
+ reference.
172
+ - Project skeleton: packaging, `fieldtrial --version`, CI and the documentation site.
173
+
174
+ [Unreleased]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0...HEAD
175
+ [0.2.0]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0...v0.2.0
176
+ [0.2.0rc1]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0a2...v0.2.0rc1
177
+ [0.2.0a2]: https://github.com/rokbenko/fieldtrial/compare/v0.2.0a1...v0.2.0a2
178
+ [0.2.0a1]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0...v0.2.0a1
179
+ [0.1.0]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0a1...v0.1.0
180
+ [0.1.0rc1]: https://github.com/rokbenko/fieldtrial/compare/v0.1.0a1...v0.1.0rc1
181
+ [0.1.0a1]: https://github.com/rokbenko/fieldtrial/releases/tag/v0.1.0a1
@@ -0,0 +1,204 @@
1
+ Metadata-Version: 2.5
2
+ Name: fieldtrial
3
+ Version: 0.2.0
4
+ Summary: Find out whether your robot policy actually got better: statistically rigorous real-world evaluation of robot policies.
5
+ Project-URL: Homepage, https://github.com/rokbenko/fieldtrial
6
+ Project-URL: Repository, https://github.com/rokbenko/fieldtrial
7
+ Project-URL: Issues, https://github.com/rokbenko/fieldtrial/issues
8
+ Project-URL: Changelog, https://github.com/rokbenko/fieldtrial/blob/main/CHANGELOG.md
9
+ Author: Rok Benko
10
+ License-Expression: Apache-2.0
11
+ License-File: LICENSE
12
+ Keywords: confidence intervals,hypothesis testing,lerobot,policy evaluation,robot learning,robotics,statistics,vla
13
+ Classifier: Development Status :: 2 - Pre-Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3 :: Only
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Scientific/Engineering
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Topic :: Scientific/Engineering :: Mathematics
25
+ Classifier: Typing :: Typed
26
+ Requires-Python: >=3.11
27
+ Requires-Dist: alembic>=1.13
28
+ Requires-Dist: fastapi>=0.115
29
+ Requires-Dist: jinja2>=3.1
30
+ Requires-Dist: matplotlib>=3.8
31
+ Requires-Dist: numpy>=1.26
32
+ Requires-Dist: pillow>=10
33
+ Requires-Dist: pydantic>=2.7
34
+ Requires-Dist: python-multipart>=0.0.9
35
+ Requires-Dist: pyyaml>=6.0
36
+ Requires-Dist: qrcode>=7.4
37
+ Requires-Dist: rich>=13.8
38
+ Requires-Dist: scipy>=1.13
39
+ Requires-Dist: sqlalchemy>=2.0.30
40
+ Requires-Dist: sse-starlette>=2.1
41
+ Requires-Dist: typer>=0.15
42
+ Requires-Dist: uvicorn[standard]>=0.30
43
+ Provides-Extra: capture
44
+ Requires-Dist: av>=12; extra == 'capture'
45
+ Requires-Dist: opencv-python-headless>=4.8; extra == 'capture'
46
+ Provides-Extra: dev
47
+ Requires-Dist: httpx>=0.28; extra == 'dev'
48
+ Requires-Dist: hypothesis>=6.130; extra == 'dev'
49
+ Requires-Dist: import-linter>=2.1; extra == 'dev'
50
+ Requires-Dist: mypy>=2.0; extra == 'dev'
51
+ Requires-Dist: pre-commit>=4.0; extra == 'dev'
52
+ Requires-Dist: pytest-cov>=6.0; extra == 'dev'
53
+ Requires-Dist: pytest>=8.3; extra == 'dev'
54
+ Requires-Dist: ruff>=0.16; extra == 'dev'
55
+ Requires-Dist: scipy-stubs>=1.15; extra == 'dev'
56
+ Requires-Dist: statsmodels>=0.14.4; extra == 'dev'
57
+ Requires-Dist: types-pyyaml>=6.0; extra == 'dev'
58
+ Provides-Extra: docs
59
+ Requires-Dist: mkdocs-material>=9.6; extra == 'docs'
60
+ Requires-Dist: mkdocs<2,>=1.6; extra == 'docs'
61
+ Requires-Dist: mkdocstrings[python]>=0.29; extra == 'docs'
62
+ Provides-Extra: lerobot
63
+ Requires-Dist: pyarrow>=15; extra == 'lerobot'
64
+ Provides-Extra: openpi
65
+ Requires-Dist: websockets>=13; extra == 'openpi'
66
+ Description-Content-Type: text/markdown
67
+
68
+ # fieldtrial
69
+
70
+ **Find out whether your robot policy actually got better.**
71
+
72
+ [![CI](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml/badge.svg)](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml)
73
+ [![PyPI](https://img.shields.io/pypi/v/fieldtrial.svg)](https://pypi.org/project/fieldtrial/)
74
+ [![License: Apache-2.0](https://img.shields.io/badge/license-Apache--2.0-blue.svg)](https://github.com/rokbenko/fieldtrial/blob/main/LICENSE)
75
+
76
+ fieldtrial is an open-source Python framework for statistically rigorous, real-world
77
+ evaluation of robot policies. LeRobot trains the policy; fieldtrial tells you whether it
78
+ actually got better.
79
+
80
+ ## Why
81
+
82
+ Real-world evaluation is slow, noisy and ad hoc. With 40 rollouts, the 95% confidence
83
+ interval for a success rate is 12 to 15 percentage points wide on each side, so many apparent
84
+ improvements from a fine-tuning run are just noise. fieldtrial helps at every stage:
85
+
86
+ - **Before:** how many rollouts do I need? What is the smallest difference this
87
+ evaluation can detect?
88
+ - **During:** a randomized, blinded schedule, and a phone-friendly operator console that
89
+ records the outcome, the furthest stage reached and the failure mode of each rollout.
90
+ - **After:** the right statistics for the design (exact tests, paired analyses,
91
+ multiplicity control), honest wording, and a self-contained report.
92
+
93
+ ## 30-second demo
94
+
95
+ ```console
96
+ $ uvx fieldtrial compare 74/80 91/120
97
+ Arm 1: 74/80 = 92.5% (95% CI 84.6%–96.5%)
98
+ Arm 2: 91/120 = 75.8% (95% CI 67.4%–82.6%)
99
+ Difference +16.7 pp, Newcombe 95% CI [+6.2 pp, +26.0 pp]
100
+ Boschloo p = 0.0020 (two-sided); rejects H0 at α = 0.05
101
+ Fisher p = 0.0022
102
+
103
+ $ uvx fieldtrial power --p1 0.76 --p2 0.90 # rollouts needed to detect 76% → 90%
104
+ 112 per arm (pooled-z)
105
+
106
+ $ uvx fieldtrial demo # a simulated study in the console
107
+ ```
108
+
109
+ `fieldtrial demo` opens a half-run, blinded study with a simulated robot. Run the rest
110
+ from the keyboard (Space, Space, Enter), unblind, and read the report.
111
+
112
+ <p>
113
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-running.png" alt="The operator console on a phone: a trial in progress with its blind code, timer and a large Stop button" width="260">
114
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-label.png" alt="Labelling a trial: furthest stage reached, why it ended, failure tags" width="260">
115
+ </p>
116
+ <p>
117
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/report.png" alt="The HTML report: summary, success rate per arm with confidence intervals, primary analysis and a forest plot" width="560">
118
+ </p>
119
+
120
+ <!-- A short GIF of a trial run in the console goes here. -->
121
+
122
+ ## Quickstart
123
+
124
+ ```console
125
+ $ uv tool install fieldtrial # or: pip install fieldtrial
126
+ $ fieldtrial init my-study # writes my-study/study.yaml
127
+ $ fieldtrial plan my-study --baseline 0.75
128
+ $ fieldtrial lock my-study # freezes the design and randomizes the schedule
129
+ $ fieldtrial serve my-study --lan # scan the QR code with a phone at the robot
130
+ $ fieldtrial unblind my-study
131
+ $ fieldtrial report my-study # my-study/reports/report.html
132
+ ```
133
+
134
+ A study is one folder: `study.yaml` (the design) and a SQLite database. Everything is
135
+ local and works offline; fieldtrial sends no telemetry.
136
+
137
+ - [Documentation](https://rokbenko.github.io/fieldtrial/): quickstart, concepts, guides
138
+ and a statistics reference.
139
+ - [Re-analysis of Dream Machines' published pi0.5 results](https://github.com/rokbenko/fieldtrial/tree/main/examples/dream-machines-pi05):
140
+ which of 30 published comparisons the data actually resolve.
141
+
142
+ ## What you get
143
+
144
+ - **Design:**
145
+ - randomized complete blocks with balanced arm order
146
+ - crossover rounds for tasks whose scene carries over
147
+ - checkpoint ladders
148
+ - planned interim looks with early stopping
149
+ - blind codes, a design hash, locking and logged amendments
150
+ - **Real blinding:** a command runner that launches your rollout command for each arm,
151
+ and an openpi router that sends each trial's requests to that arm's policy server.
152
+ - **Console:** sessions with a rig checklist, a timer, stage and failure-tag labels,
153
+ 10-second undo, invalid trials with automatic rescheduling, a live mirror screen,
154
+ keyboard and foot-pedal keys, LAN access with a QR code.
155
+ - **Evidence:** an evaluation camera that records every trial, rig checks against a
156
+ reference photo, and links from trials to LeRobot dataset episodes (with DAgger
157
+ interventions).
158
+ - **Analysis:** the primary test follows from the locked design:
159
+ - exact McNemar with a Tango interval, Cochran–Mantel–Haenszel, or Cochran's Q with Holm
160
+ - group-sequential boundaries
161
+ - the association of success with training step, and plateau detection
162
+ - a period-adjusted crossover test
163
+
164
+ Every analysis also gets an independent-samples sensitivity analysis, stage funnels,
165
+ time to success, drift checks and a list of every deviation from the plan.
166
+ - **Reports:** self-contained HTML with charts, Markdown for pull requests, and a
167
+ versioned JSON results model.
168
+ - **Integration:** a REST API with a dependency-free Python client for custom runtimes,
169
+ CSV import and export, and `fieldtrial.stats` as a library.
170
+
171
+ Every statistical function is tested against an independent reference implementation.
172
+ Reports only describe a difference when the pre-registered test rejects; otherwise they
173
+ say what the study could have detected.
174
+
175
+ ## How it fits with LeRobot and openpi
176
+
177
+ fieldtrial complements LeRobot and openpi and never forks them. It is not a training
178
+ framework, a simulation benchmark, a robot driver, a labeling platform or a cloud
179
+ service. In manual mode, fieldtrial schedules and records the trials and you run the
180
+ robot however you like. For real blinding, the `command` runner launches your rollout
181
+ command (for example `lerobot-rollout`) for each arm, and the `openpi_router` runner
182
+ routes an openpi client's traffic to each trial's policy server; see
183
+ [Real blinding with runners](https://rokbenko.github.io/fieldtrial/guides/runners/).
184
+
185
+ ## How to cite
186
+
187
+ If fieldtrial helps your research, please cite it (see
188
+ [CITATION.cff](https://github.com/rokbenko/fieldtrial/blob/main/CITATION.cff)):
189
+
190
+ ```bibtex
191
+ @software{benko_fieldtrial,
192
+ author = {Benko, Rok},
193
+ title = {fieldtrial: statistically rigorous real-world evaluation for robot policies},
194
+ url = {https://github.com/rokbenko/fieldtrial},
195
+ license = {Apache-2.0}
196
+ }
197
+ ```
198
+
199
+ For evaluation practice in general, see Kress-Gazit et al. (2024), *Robot Learning as an
200
+ Empirical Science: Best Practices for Policy Evaluation*, arXiv:2409.09491.
201
+
202
+ ## Development
203
+
204
+ See [CONTRIBUTING.md](https://github.com/rokbenko/fieldtrial/blob/main/CONTRIBUTING.md).
@@ -0,0 +1,137 @@
1
+ # fieldtrial
2
+
3
+ **Find out whether your robot policy actually got better.**
4
+
5
+ [![CI](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml/badge.svg)](https://github.com/rokbenko/fieldtrial/actions/workflows/ci.yml)
6
+ [![PyPI](https://img.shields.io/pypi/v/fieldtrial.svg)](https://pypi.org/project/fieldtrial/)
7
+ [![License: Apache-2.0](https://img.shields.io/badge/license-Apache--2.0-blue.svg)](https://github.com/rokbenko/fieldtrial/blob/main/LICENSE)
8
+
9
+ fieldtrial is an open-source Python framework for statistically rigorous, real-world
10
+ evaluation of robot policies. LeRobot trains the policy; fieldtrial tells you whether it
11
+ actually got better.
12
+
13
+ ## Why
14
+
15
+ Real-world evaluation is slow, noisy and ad hoc. With 40 rollouts, the 95% confidence
16
+ interval for a success rate is 12 to 15 percentage points wide on each side, so many apparent
17
+ improvements from a fine-tuning run are just noise. fieldtrial helps at every stage:
18
+
19
+ - **Before:** how many rollouts do I need? What is the smallest difference this
20
+ evaluation can detect?
21
+ - **During:** a randomized, blinded schedule, and a phone-friendly operator console that
22
+ records the outcome, the furthest stage reached and the failure mode of each rollout.
23
+ - **After:** the right statistics for the design (exact tests, paired analyses,
24
+ multiplicity control), honest wording, and a self-contained report.
25
+
26
+ ## 30-second demo
27
+
28
+ ```console
29
+ $ uvx fieldtrial compare 74/80 91/120
30
+ Arm 1: 74/80 = 92.5% (95% CI 84.6%–96.5%)
31
+ Arm 2: 91/120 = 75.8% (95% CI 67.4%–82.6%)
32
+ Difference +16.7 pp, Newcombe 95% CI [+6.2 pp, +26.0 pp]
33
+ Boschloo p = 0.0020 (two-sided); rejects H0 at α = 0.05
34
+ Fisher p = 0.0022
35
+
36
+ $ uvx fieldtrial power --p1 0.76 --p2 0.90 # rollouts needed to detect 76% → 90%
37
+ 112 per arm (pooled-z)
38
+
39
+ $ uvx fieldtrial demo # a simulated study in the console
40
+ ```
41
+
42
+ `fieldtrial demo` opens a half-run, blinded study with a simulated robot. Run the rest
43
+ from the keyboard (Space, Space, Enter), unblind, and read the report.
44
+
45
+ <p>
46
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-running.png" alt="The operator console on a phone: a trial in progress with its blind code, timer and a large Stop button" width="260">
47
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/console-label.png" alt="Labelling a trial: furthest stage reached, why it ended, failure tags" width="260">
48
+ </p>
49
+ <p>
50
+ <img src="https://raw.githubusercontent.com/rokbenko/fieldtrial/main/docs/assets/report.png" alt="The HTML report: summary, success rate per arm with confidence intervals, primary analysis and a forest plot" width="560">
51
+ </p>
52
+
53
+ <!-- A short GIF of a trial run in the console goes here. -->
54
+
55
+ ## Quickstart
56
+
57
+ ```console
58
+ $ uv tool install fieldtrial # or: pip install fieldtrial
59
+ $ fieldtrial init my-study # writes my-study/study.yaml
60
+ $ fieldtrial plan my-study --baseline 0.75
61
+ $ fieldtrial lock my-study # freezes the design and randomizes the schedule
62
+ $ fieldtrial serve my-study --lan # scan the QR code with a phone at the robot
63
+ $ fieldtrial unblind my-study
64
+ $ fieldtrial report my-study # my-study/reports/report.html
65
+ ```
66
+
67
+ A study is one folder: `study.yaml` (the design) and a SQLite database. Everything is
68
+ local and works offline; fieldtrial sends no telemetry.
69
+
70
+ - [Documentation](https://rokbenko.github.io/fieldtrial/): quickstart, concepts, guides
71
+ and a statistics reference.
72
+ - [Re-analysis of Dream Machines' published pi0.5 results](https://github.com/rokbenko/fieldtrial/tree/main/examples/dream-machines-pi05):
73
+ which of 30 published comparisons the data actually resolve.
74
+
75
+ ## What you get
76
+
77
+ - **Design:**
78
+ - randomized complete blocks with balanced arm order
79
+ - crossover rounds for tasks whose scene carries over
80
+ - checkpoint ladders
81
+ - planned interim looks with early stopping
82
+ - blind codes, a design hash, locking and logged amendments
83
+ - **Real blinding:** a command runner that launches your rollout command for each arm,
84
+ and an openpi router that sends each trial's requests to that arm's policy server.
85
+ - **Console:** sessions with a rig checklist, a timer, stage and failure-tag labels,
86
+ 10-second undo, invalid trials with automatic rescheduling, a live mirror screen,
87
+ keyboard and foot-pedal keys, LAN access with a QR code.
88
+ - **Evidence:** an evaluation camera that records every trial, rig checks against a
89
+ reference photo, and links from trials to LeRobot dataset episodes (with DAgger
90
+ interventions).
91
+ - **Analysis:** the primary test follows from the locked design:
92
+ - exact McNemar with a Tango interval, Cochran–Mantel–Haenszel, or Cochran's Q with Holm
93
+ - group-sequential boundaries
94
+ - the association of success with training step, and plateau detection
95
+ - a period-adjusted crossover test
96
+
97
+ Every analysis also gets an independent-samples sensitivity analysis, stage funnels,
98
+ time to success, drift checks and a list of every deviation from the plan.
99
+ - **Reports:** self-contained HTML with charts, Markdown for pull requests, and a
100
+ versioned JSON results model.
101
+ - **Integration:** a REST API with a dependency-free Python client for custom runtimes,
102
+ CSV import and export, and `fieldtrial.stats` as a library.
103
+
104
+ Every statistical function is tested against an independent reference implementation.
105
+ Reports only describe a difference when the pre-registered test rejects; otherwise they
106
+ say what the study could have detected.
107
+
108
+ ## How it fits with LeRobot and openpi
109
+
110
+ fieldtrial complements LeRobot and openpi and never forks them. It is not a training
111
+ framework, a simulation benchmark, a robot driver, a labeling platform or a cloud
112
+ service. In manual mode, fieldtrial schedules and records the trials and you run the
113
+ robot however you like. For real blinding, the `command` runner launches your rollout
114
+ command (for example `lerobot-rollout`) for each arm, and the `openpi_router` runner
115
+ routes an openpi client's traffic to each trial's policy server; see
116
+ [Real blinding with runners](https://rokbenko.github.io/fieldtrial/guides/runners/).
117
+
118
+ ## How to cite
119
+
120
+ If fieldtrial helps your research, please cite it (see
121
+ [CITATION.cff](https://github.com/rokbenko/fieldtrial/blob/main/CITATION.cff)):
122
+
123
+ ```bibtex
124
+ @software{benko_fieldtrial,
125
+ author = {Benko, Rok},
126
+ title = {fieldtrial: statistically rigorous real-world evaluation for robot policies},
127
+ url = {https://github.com/rokbenko/fieldtrial},
128
+ license = {Apache-2.0}
129
+ }
130
+ ```
131
+
132
+ For evaluation practice in general, see Kress-Gazit et al. (2024), *Robot Learning as an
133
+ Empirical Science: Best Practices for Policy Evaluation*, arXiv:2409.09491.
134
+
135
+ ## Development
136
+
137
+ See [CONTRIBUTING.md](https://github.com/rokbenko/fieldtrial/blob/main/CONTRIBUTING.md).