outerloop-science 0.1.1__tar.gz → 0.2.0rc1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (187) hide show
  1. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/CHANGELOG.md +48 -0
  2. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/CITATION.cff +1 -1
  3. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/PKG-INFO +1 -1
  4. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/agent-substrate.md +4 -4
  5. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/architecture.md +14 -9
  6. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/base-reintegration.md +7 -7
  7. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/consolidation.md +1 -1
  8. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/dispatcher.md +5 -5
  9. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/headline.md +3 -5
  10. outerloop_science-0.2.0rc1/docs/design/lifecycle.md +338 -0
  11. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/orchestrator-verify.md +13 -79
  12. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/research-lines.md +6 -1
  13. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/research-loop-buildout.md +23 -1
  14. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/research-loop.md +4 -3
  15. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/reviewer-infra.md +4 -2
  16. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/roles.md +9 -8
  17. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/install.md +82 -5
  18. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/roadmap.md +1 -1
  19. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/tick_chain.sbatch +1 -1
  20. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/__init__.py +1 -1
  21. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/appauth.py +4 -2
  22. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/appmanifest.py +7 -0
  23. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/attempt.py +1688 -827
  24. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/brief.py +37 -100
  25. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/cli.py +81 -0
  26. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/climbboard.py +40 -17
  27. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/compute.py +9 -1
  28. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/contract.py +26 -3
  29. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/dispatch.py +10 -0
  30. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/github.py +59 -36
  31. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/harness.py +7 -1
  32. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/housekeeping.py +2 -3
  33. outerloop_science-0.2.0rc1/src/outerloop/inbox.py +581 -0
  34. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/init.py +131 -43
  35. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/intake.py +43 -16
  36. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/measure.py +5 -7
  37. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/orchestrator.py +286 -147
  38. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/panel.py +4 -20
  39. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/roles.py +0 -38
  40. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/rolespec.py +1 -3
  41. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/runstate.py +134 -39
  42. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/steward.py +31 -25
  43. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/syscall.py +116 -153
  44. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/syscall_cli.py +119 -23
  45. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/tick.py +306 -463
  46. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/verifier.py +2 -2
  47. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/watcher.py +1 -1
  48. outerloop_science-0.2.0rc1/tests/test_app_permissions.py +345 -0
  49. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_appauth.py +27 -0
  50. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_appmanifest.py +3 -0
  51. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_attempt.py +832 -196
  52. outerloop_science-0.2.0rc1/tests/test_attempt_review.py +1292 -0
  53. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_bot_aliases.py +1 -1
  54. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_bot_login.py +2 -2
  55. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_brief.py +39 -27
  56. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_climbboard.py +54 -14
  57. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_compute.py +8 -0
  58. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_contract.py +29 -0
  59. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_github.py +102 -75
  60. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_harness.py +15 -2
  61. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_housekeeping.py +3 -3
  62. outerloop_science-0.2.0rc1/tests/test_inbox.py +610 -0
  63. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_init.py +40 -11
  64. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_intake.py +30 -0
  65. outerloop_science-0.2.0rc1/tests/test_lifecycle_cli.py +23 -0
  66. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_local_compute.py +46 -23
  67. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_orchestrator.py +185 -94
  68. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_panel.py +16 -5
  69. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_role_runner.py +0 -11
  70. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_runstate.py +130 -5
  71. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_steward.py +18 -2
  72. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_syscall.py +87 -125
  73. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_syscall_cli.py +114 -3
  74. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_tick.py +803 -1274
  75. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_watcher.py +1 -3
  76. outerloop_science-0.1.1/src/outerloop/followup.py +0 -2172
  77. outerloop_science-0.1.1/tests/test_followup.py +0 -3128
  78. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/.gitignore +0 -0
  79. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/.pre-commit-config.yaml +0 -0
  80. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/.python-version +0 -0
  81. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/CLAUDE.md +0 -0
  82. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/CONTRIBUTING.md +0 -0
  83. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/LICENSE +0 -0
  84. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/NOTICE +0 -0
  85. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/README.md +0 -0
  86. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/RELEASING.md +0 -0
  87. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/SECURITY.md +0 -0
  88. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/containers/README.md +0 -0
  89. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/containers/agent-py312.def +0 -0
  90. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/assets/icon-dark.svg +0 -0
  91. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/assets/icon-light.svg +0 -0
  92. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/assets/icon.svg +0 -0
  93. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/community.md +0 -0
  94. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/compute.md +0 -0
  95. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/contract.md +0 -0
  96. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/eval-cache.md +0 -0
  97. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/external.md +0 -0
  98. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/github-app-auth.md +0 -0
  99. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/judge-placement.md +0 -0
  100. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/meta.md +0 -0
  101. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/onboarding.md +0 -0
  102. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/public-surface.md +0 -0
  103. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/resident-tick.md +0 -0
  104. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/review-placement.md +0 -0
  105. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/role-cli.md +0 -0
  106. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/scaling.md +0 -0
  107. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/design/session-watcher.md +0 -0
  108. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/reviewer.md +0 -0
  109. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/docs/validation/author-syscalls.md +0 -0
  110. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/examples/review.yml +0 -0
  111. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/pyproject.toml +0 -0
  112. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/README.md +0 -0
  113. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/install_codex.sh +0 -0
  114. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/install_hermes.sh +0 -0
  115. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/requeue_moved_successors.sh +0 -0
  116. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/setup_branch_protection.sh +0 -0
  117. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/sweep_git_locks.sh +0 -0
  118. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/tick_deploy.sh +0 -0
  119. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/scripts/tick_resident.sh +0 -0
  120. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/__main__.py +0 -0
  121. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/contract_cli.py +0 -0
  122. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/disk.py +0 -0
  123. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/evalcache.py +0 -0
  124. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/image.py +0 -0
  125. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/launchlog.py +0 -0
  126. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/limits.py +0 -0
  127. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/maintain.py +0 -0
  128. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/maintain_agent_cli.py +0 -0
  129. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/maintain_post_cli.py +0 -0
  130. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/markers.py +0 -0
  131. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/paths.py +0 -0
  132. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/posting.py +0 -0
  133. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/progress.py +0 -0
  134. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/py.typed +0 -0
  135. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/review.py +0 -0
  136. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/review_agent.py +0 -0
  137. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/review_agent_cli.py +0 -0
  138. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/review_post_cli.py +0 -0
  139. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/review_summarize_cli.py +0 -0
  140. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/role_runner.py +0 -0
  141. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/style.py +0 -0
  142. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/verify_agent.py +0 -0
  143. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/verify_agent_cli.py +0 -0
  144. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/src/outerloop/verify_post_cli.py +0 -0
  145. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/conftest.py +0 -0
  146. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/fakes.py +0 -0
  147. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/helpers.py +0 -0
  148. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_channel_dir.py +0 -0
  149. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_codex_harness.py +0 -0
  150. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_contract_cli.py +0 -0
  151. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_contract_names.py +0 -0
  152. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_disk.py +0 -0
  153. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_dispatch.py +0 -0
  154. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_env_bridge.py +0 -0
  155. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_evalcache.py +0 -0
  156. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_hardening.py +0 -0
  157. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_hermes_harness.py +0 -0
  158. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_image.py +0 -0
  159. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_import.py +0 -0
  160. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_launchlog.py +0 -0
  161. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_limits.py +0 -0
  162. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_maintain.py +0 -0
  163. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_markers.py +0 -0
  164. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_measure.py +0 -0
  165. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_measure_and_decide.py +0 -0
  166. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_packaging.py +0 -0
  167. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_paths.py +0 -0
  168. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_posting.py +0 -0
  169. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_progress.py +0 -0
  170. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_requeue_moved_successors.py +0 -0
  171. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review.py +0 -0
  172. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review_agent.py +0 -0
  173. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review_agent_cli.py +0 -0
  174. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review_hardening.py +0 -0
  175. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review_policy.py +0 -0
  176. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_review_summarize.py +0 -0
  177. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_rolespec.py +0 -0
  178. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_start.py +0 -0
  179. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_sweep_git_locks.py +0 -0
  180. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_tick_chain_successors.py +0 -0
  181. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_tick_resident.py +0 -0
  182. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_tiers.py +0 -0
  183. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_verifier.py +0 -0
  184. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_verify_agent.py +0 -0
  185. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_verify_agent_cli.py +0 -0
  186. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/tests/test_version.py +0 -0
  187. {outerloop_science-0.1.1 → outerloop_science-0.2.0rc1}/uv.lock +0 -0
@@ -6,6 +6,54 @@ Versions follow [SemVer](https://semver.org).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ### Fixed
10
+
11
+ - `outerloop permissions` checks App access and opens the next page to edit or accept missing permissions with `--open`; upgrades, starts and sweep warnings provide guidance for existing installations.
12
+
13
+ - In auto mode the sweep merges only a clean PR at the head the kernel measured and approved. Publish never arms GitHub auto-merge; the sweep withdraws old arms and reports head moves to the author.
14
+
15
+ - A reply returning after PR merge or close preserves the run’s ending. GitHub outages no longer skip job and deadline handling, failed reply posts suppress duplicate final text, and terminal snapshot cleanup survives notebook failures. Inbox delivery retries refused messages and refuses symlinked destinations.
16
+
17
+ - A review panel skipped at preflight or in the follow-up now sends the author a kernel note and records the reason in the submit’s PR addendum. These publishes never arm auto-merge.
18
+
19
+ - A submitted review measurement could go unreported when its push failed. The number is now posted once, as soon as it is known. Review submits push a bot commit with the report’s first line and measured number as its message.
20
+ - On local compute an author-sleep wake could lose an improvement: the gate answers inline there, and the wake only knew how to end a run, so the result became an aborted ending with no PR (cluster0, three runs). Fresh and resumed climbs now share one terminal: the report, the line notebook, a PR or an ending record, and the issue note. A wake that opened its PR but died before recording it reconciles to that PR instead of opening a second, and a park's snapshot is released only once the run has left waiting.
21
+ - A research line lost its snapshot when the author's session reset its branch to main: the seal parented on main, the push was refused, and the run left no notebook entry (#368). The kernel now records the line head itself and the line only moves forward from that record. Memory files a reset dropped from the tree come back at the seal; files the session deleted on the line stay deleted.
22
+ - Intake dropped an issue from a private org member without a word: the App's token sees such an author as CONTRIBUTOR, and only OWNER/MEMBER/COLLABORATOR qualified. Every skipped issue is now logged with its reason, a maintainer's `outerloop:task` label vouches for an issue regardless of association, and new Apps request members read so private members read as MEMBER.
23
+
24
+ ### Changed
25
+
26
+ - Completed CI checks reach the author as messages with bounded log tails. Failed checks wake idle runs; successful, neutral and skipped checks wait as context.
27
+
28
+ - Messages carry an origin and a qualified repository thread. Staged replies keep their destination through retries and later PR changes.
29
+
30
+ - Memory guidance now covers every session boundary and asks authors to record durable findings as they learn them. Sleep, submit and end help say the session ends there.
31
+
32
+ - The board's live strip says what a parked run waits on beside its state: jobs, gate, review or wake. Read off the record; the three states are unchanged.
33
+ - Runs have three states: running, parked and ended; the board and the logs show those names, and an open PR is a link on the run. The sweep delivers comments, base moves and job results through one wake path; the separate follow-up job is gone. Old state names are mapped on read; inbox cursors migrate once under the run lease, and every tick logs the number of legacy follow-up records until it is zero. An older kernel cannot read the new state names; that is the only incompatible field, and fleets are updated by commit. The contract's `followup_job_minutes` now sizes the wake job of a run with an open PR.
34
+
35
+
36
+ - Authors can stage `end [--report <file>]` and end their turn: without a PR it ends as a negative result with the last failed verdict's note, or “ended without a submit”; with an open PR it posts the supplied report and parks until a message arrives. End costs nothing and cannot accompany launch, sleep or submit.
37
+ - Opening a run's first PR grants a review top-up once: 2 launches, 4 sleeps and 0.5 GPU-hours by default, configurable through `budgets.review_topup`. The increased ceilings and prior spend appear in the tool, brief and every wake.
38
+
39
+ - Submit works in review and fast-forwards the PR after a confirmed auto-merge disarm; the measured number is posted first, even if publication is refused. Each review leg measures against its freshly fetched base. Verdicts, findings and publish refusals reach the author as messages. A reply staged during a review leg suppresses its final-text comment. A failed submitted park ends as negative-result only when no author session can resume to receive its verdict; its report and notebook are saved and its issue claim released. Sessions without the tool (no launcher or no resume support) are still measured at finish and may end on the verdict by design. Submit needs no prior launch or report. A session offered submit that stops without it ends unmeasured, or returns to review if it has a PR. Legacy follow-up re-measures retire on their next wake and release their snapshots.
40
+
41
+ - Authors can post replies through `reply` and launch experiments or sleep while a PR is in review, using the run’s remaining budget. Comments and base moves received while parked reach the author at its next wake. Replies are kept in an outbox for retries, review launches check committed edits, and closing a run holds its wake lease. Review edits are measured and published only on submit, using the same remaining GPU budget as other author work.
42
+
43
+ - Wake messages now use one inbox and one renderer. Launch results, gate verdicts, panel findings, review comments and base moves reach the session in the same fenced format, from files kept beside the run's record. Advisory panel findings now reach the author alongside blocking ones. The sibling view is refreshed at every wake of a parked author session. The kernel's wake text states facts; the research advice it used to carry is gone.
44
+
45
+ - Dispatched wakes are on by default. The old on-switch (`OUTERLOOP_DISPATCH_WAKE=1` or a `DISPATCH_WAKE` sentinel) is gone; the operator turns wakes off with `OUTERLOOP_DISPATCH_WAKE=0` or a `<root>/DISARM_WAKE` sentinel, and a dry sweep says so in the log. An unarmed loop stranded every parked run silently, which local compute, able to park since 0.1.2, hit at once.
46
+ - The brief's launch section says that a launch runs a sealed snapshot of the working tree and is scope-checked like the final tree, so experiment scripts must live under the contract's allowed paths.
47
+
48
+ - Whether jobs need a lane is now the compute backend's word (`has_lanes`): the measurer, the launcher, and the tick's GPU preflight ask it instead of testing for local mode. The dispatch settings always exist (a backend always does), so local compute arms the author's `launch`/`sleep` syscalls for benchmarks with `depth_k`, and a local wake no longer needs a container image. A `--image` path that is not a file is refused at the command line instead of silently degrading the run to inline evaluation.
49
+ - Author sessions start with no GPU visible on any backend (`CUDA_VISIBLE_DEVICES` is empty), so experiments that need one go through `launch`, where they are recorded in the ledger and metered against the GPU budget. This is the bare session's default; a contained session has no GPU device at all, which is the enforcement.
50
+
51
+ ## [0.1.2] - 2026-09-11
52
+
53
+ ### Fixed
54
+
55
+ - A GPU benchmark on local compute aborted at its first gate measure with "no GPU lane is configured". The measure placement now follows the launch placement's rule: local compute has no lanes, the job runs on the machine's own GPUs.
56
+
9
57
  ## [0.1.1] - 2026-09-11
10
58
 
11
59
  ### Fixed
@@ -13,5 +13,5 @@ authors:
13
13
  repository-code: "https://github.com/outerloop-science/outerloop"
14
14
  url: "https://outerloop.science"
15
15
  license: Apache-2.0
16
- version: 0.1.1
16
+ version: 0.1.2
17
17
  date-released: "2026-09-11"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: outerloop-science
3
- Version: 0.1.1
3
+ Version: 0.2.0rc1
4
4
  Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
5
5
  Project-URL: Homepage, https://outerloop.science
6
6
  Project-URL: Repository, https://github.com/outerloop-science/outerloop
@@ -61,7 +61,7 @@ per tiny thing is the accretion trap in a new costume.
61
61
  - steward: `ruler-hardening` (when/how to make the metric harder once it's gamed —
62
62
  add a transfer split, harder cells) · `benchmark-design` (noise floors, seeds,
63
63
  why a sweep beats a single point)
64
- - follow-up: `respond-to-review` (address maintainer comments; which task
64
+ - resumed author: `respond-to-review` (address maintainer comments; which task
65
65
  instruction a wake supersedes; honest scope)
66
66
 
67
67
  **Target (e.g. `yolo-jepa`)**
@@ -138,7 +138,7 @@ planned set — none are wired yet):
138
138
  | author | Read/Grep/Glob/Write/Edit/Bash | kernel-primer, plain-style, hypothesis-discipline, honest-method, experiment-lifecycle, research-report (+ self-review, analyze-results — proposed) | literature-search (retriever), result-aggregation |
139
139
  | reviewer | Read/Grep/Glob + pr-context-read, retriever | kernel-primer, plain-style, review-rubric, read-only-investigation | evidence-sweep (parallel read-only file/caller sweeps on large diffs), reference-check (retriever) |
140
140
  | verifier | Read/Grep/Glob + pr-context-read, retriever | kernel-primer, plain-style, integrity-lens, read-only-investigation | evidence-sweep (read-only; e.g. trace every consumer of a changed ruler input) |
141
- | followup | editing set, resuming role's key/scope | kernel-primer, plain-style, respond-to-review | inherits the resumed role's |
141
+ | resumed author | editing set, resuming role's key/scope | kernel-primer, plain-style, respond-to-review | inherits the resumed role's |
142
142
  | steward | editing set, own territory | kernel-primer, plain-style, ruler-hardening, benchmark-design | literature-search (eval conventions) |
143
143
 
144
144
  The self-review skill's relationship to the panel is specified in
@@ -252,7 +252,7 @@ is the rare exception).
252
252
 
253
253
  | Field | Meaning |
254
254
  | --- | --- |
255
- | `name` | role id (author, reviewer, verifier, steward, followup) |
255
+ | `name` | role id (author, reviewer, verifier, steward) |
256
256
  | `instructions` | standing role prompt, composed from skills |
257
257
  | `skills` | skill ids to load (global + role + target) |
258
258
  | `tools` | allowed tools (native + harness-provided) |
@@ -267,7 +267,7 @@ is the rare exception).
267
267
 
268
268
  Replaces the five per-role drivers' session dispatch — all five roles now run
269
269
  through `run_role` with their RoleSpec (`review_cli` and `verifier_cli` are
270
- deleted; author, follow-up, and steward dispatch from their kernel modules).
270
+ deleted; author and steward dispatch from their kernel modules).
271
271
  Kernel code; it calls into the agentic realm at step 2, and everything
272
272
  trust-critical is deterministic.
273
273
 
@@ -8,7 +8,8 @@ review.
8
8
  A background agent that co-develops the lab's benchmark-bearing repos (jepa-agent,
9
9
  egolearn): picks work from their roadmaps and benchmark gaps, implements on a
10
10
  branch, runs GPU experiments, opens a PR when a metric improves, reviews PRs, and
11
- reports weekly. Humans keep the merge button.
11
+ reports weekly. Humans keep the merge button unless a repo owner opts in to
12
+ `merge: auto` in the contract.
12
13
 
13
14
  ## Decisions
14
15
 
@@ -180,12 +181,12 @@ email, no command grammar to learn or to secure.
180
181
  its own.
181
182
 
182
183
  **In flight — humans steer through the PR.** A run whose PR is open enters
183
- `in-review` and stays alive: each tick, the sweep checks the PR for new
184
+ `parked` and stays alive: each tick, the sweep checks the PR for new
184
185
  comments by org members (the advisory reviewer never comments on bot PRs, so
185
186
  no bot-to-bot loop can form). A qualifying comment wakes the run — the same
186
187
  resume mechanism as experiment wakes, comment text data-fenced, task-level
187
- supersession only — and the agent pushes fixes and replies on the thread via
188
- the bot identity. Review-response sessions are bounded like everything else
188
+ supersession only. The author replies on the thread and submits changes
189
+ through the gate and publish. Comments wait while launched jobs are active. Review-response sessions are bounded like everything else
189
190
  (attempt counter, budget). Non-member comments never trigger a wake.
190
191
 
191
192
  **Death — a run ends in exactly one of six ways, each producing a report:**
@@ -209,7 +210,7 @@ honest wording, rather than waiting for the orphan-reconciliation pass to
209
210
  mislabel a deliberate "no" as a crash.
210
211
 
211
212
  After the report is distilled into `lessons/`, the workspace and per-run HOME
212
- are garbage-collected (grace period first — an `in-review` run's context must
213
+ are garbage-collected (grace period first — a `parked` run's context must
213
214
  survive until its PR closes). The notebook is the memory; the run directory
214
215
  is scaffolding.
215
216
 
@@ -363,11 +364,11 @@ failure can strand a run:
363
364
  strand the run.
364
365
  3. *Backup — the tick sweeps.* The tick chain (independently kept alive:
365
366
  two queued successors, heartbeat, GH-Actions watchdog) scans every run in
366
- `waiting` each tick: experiment job terminal per `sacct` + no wake lease
367
+ `parked` each tick: experiment job terminal per `sacct` + no wake lease
367
368
  or completion within a grace window → the tick dispatches the wake
368
369
  itself. This covers a lost wake job, a wake killed mid-session, and Slurm
369
370
  controller restarts that drop pending jobs.
370
- 4. *Deadline floor.* Every `waiting` run records
371
+ 4. *Deadline floor.* Every `parked` run records
371
372
  `deadline = submit_time + walltime + slack` (recomputed from `start_time`
372
373
  once the job starts, so late scheduling never truncates a healthy run).
373
374
  Past the deadline the sweep consults `sacct` and acts on what it finds:
@@ -537,5 +538,9 @@ not yet build-gating.
537
538
  - GitHub App (revisit if PAT limits bite)
538
539
  - Cloud compute backend (burst valve when Torch queues block)
539
540
  - Third-party / self-hosted models
540
- - Any form of auto-merge on code — never. The sole exception is the prose-only
541
- notebook repo.
541
+ - Auto-merge on code is off by default and is never the kernel's decision.
542
+ A repo owner may turn it on with `merge: auto` in the contract: a PR whose
543
+ gate and panel read are clean is merged by the kernel sweep, at the blessed
544
+ head only, through the repo's required checks and branch protection. The
545
+ kernel never arms GitHub auto-merge in this mode and withdraws old arms.
546
+ The prose-only notebook repo merges on its secret-scan check alone.
@@ -23,8 +23,8 @@ Two gaps produced this:
23
23
  arbitrarily far from main as siblings merge, and `research-lines.md` already
24
24
  names the danger ("a stale line reverting others' wins").
25
25
  The kernel already reconciles a PR whose base moves AFTER it opens: an open PR
26
- keeps its run in the in-review state, and the tick's follow-up conflict wake
27
- (`followup.py`) fetches the moved base into the workspace and asks the agent to
26
+ keeps its run in the parked state, and the tick's base-move inbox message
27
+ (`attempt.resume_run`) fetches the moved base into the workspace and delivers a base-move message; the author decides whether to
28
28
  merge and re-measure. So the only real gap is the first bullet — a line that
29
29
  never re-syncs main mid-run and opens its PR against a base superseded days
30
30
  earlier.
@@ -66,8 +66,8 @@ The semantic choices, settled — the mechanics follow from them.
66
66
  beaten (e.g. "warmdown was taken further by #10"); the agent decides to
67
67
  pivot or push on. The kernel never kills a line for it.
68
68
  4. **Healing a PR after its base moves: already handled, no new mechanism.**
69
- An open PR's run stays in-review, and the follow-up conflict wake already
70
- fetches the moved base and asks the agent to merge and re-measure. The
69
+ An open PR's run stays parked, and the base-move inbox message already
70
+ fetches the moved base and delivers a base-move message; the author decides whether to merge and re-measure. The
71
71
  note's earlier "post-terminal" framing was wrong (terra, #341): the run is
72
72
  not terminated while its PR is open. Base reintegration is therefore stage 1
73
73
  alone — closing the during-run drift; the post-open case needs nothing new.
@@ -77,7 +77,7 @@ The semantic choices, settled — the mechanics follow from them.
77
77
 
78
78
  **Build order.** The whole feature is one stage: re-pin at wake with the digest
79
79
  (decisions 1–3), the re-measure paid only on keep (decision 5). Decision 4 needs
80
- no code — the existing follow-up conflict wake already covers a base that moves
80
+ no code — the existing base-move inbox message already covers a base that moves
81
81
  after the PR opens.
82
82
 
83
83
  ## Why this shape addresses the problem
@@ -86,7 +86,7 @@ The base pin is correct during a run — you cannot measure improvement against
86
86
  a moving target. The failure is not the pin; it is holding one pin for a
87
87
  multi-day run and having no reconciliation once the run ends. Waking the line to merge the fresh base keeps the stability the pin gives
88
88
  while bounding the drift to a single iteration, and it reuses the exact
89
- mechanism the in-review conflict wake already uses — the kernel fetches, the
89
+ mechanism the parked PR wake already uses — the kernel fetches, the
90
90
  agent merges — rather than adding new kernel git machinery. The post-open case
91
- needs nothing more: an open PR's run stays in-review and that same conflict
91
+ needs nothing more: an open PR's run stays parked and that same conflict
92
92
  wake already reconciles it.
@@ -120,7 +120,7 @@ scope, key), and their spend counts against its budget.
120
120
 
121
121
  | Now | Fate |
122
122
  | --- | --- |
123
- | `climb`, `steward`, `followup` | Session dispatch collapsed: all five roles run through **one role-runner + a RoleSpec** (judges via their agent modules, `review_cli`/`verifier_cli` deleted; author via `climb_once`; follow-up via `respond_once` under the resuming role's key and scope; steward via `live_steward`) with the harness built from the spec (one `build_harness` for every role). What remains of these modules is each role's kernel half — measure/gate/PR/wake plumbing. The brief/skills halves becoming app config is the remaining consolidation. |
123
+ | `climb`, `steward`, `wake` | Session dispatch collapsed: all five roles run through **one role-runner + a RoleSpec** (judges via their agent modules, `review_cli`/`verifier_cli` deleted; author via `climb_once`; follow-up via `attempt.run_author_leg` under the resuming role's key and scope; steward via `live_steward`) with the harness built from the spec (one `build_harness` for every role). What remains of these modules is each role's kernel half — measure/gate/PR/wake plumbing. The brief/skills halves becoming app config is the remaining consolidation. |
124
124
  | `review`, `verifier` (rendering, verdict/blocking machinery) | On the agent-session path; hold the shared vocabulary and rendering both judges use. |
125
125
  | `harness`, `brief` | The seam. Keep. Adapters live: claude, codex, hermes. |
126
126
  | `orchestrator`, `contract`, `github`, `compute`, `runstate`, `disk`, `limits`, `intake` | Kernel. Barely moves — the point. |
@@ -35,7 +35,7 @@ Dispatch is a **syscall on the existing compute seam** — `submit(JobSpec)` +
35
35
  an `afterany` wake job, the same two primitives the tick chain already uses,
36
36
  zero new kernel concepts. When the orchestrator (or, in phase 2, an author
37
37
  session through a tool) needs an experiment run, it submits the job, writes
38
- the run record into `waiting` with `experiment_job_id` and a deadline, and
38
+ the run record into `parked` with `experiment_job_id` and a deadline, and
39
39
  **ends the session**. Results arrive by wake: the afterany job is the
40
40
  primary delivery, the tick's waiting-sweep is the backup (both exist; the
41
41
  sweep runs dry today). Nothing polls, and no clock runs while a job queues.
@@ -160,7 +160,7 @@ AFTER the session:
160
160
  3. The MEASURE-AND-DECIDE phase — a pure function of `(base_sha,
161
161
  candidate_sha, contract, seed, suite_seed)` — dispatches its measures
162
162
  through the `Measurer`, and on `MeasurementPending` the run parks as
163
- `waiting`. It measures LAZILY, in the same order `climb_once` did: first
163
+ `parked`. It measures LAZILY, in the same order `climb_once` did: first
164
164
  baseline@base_sha + candidate@candidate_sha (one wake); only if that pair
165
165
  clears the improvement threshold AND the diff touched shared code does it
166
166
  dispatch the sibling pairs (a second wake). A non-improving candidate never
@@ -198,7 +198,7 @@ Sub-parts, each its own PR through the panel:
198
198
  logic behind a re-enterable function over committed shas; pure, tested
199
199
  with a fake measurer.
200
200
  - **B.2b:** wire `live_climb` — session -> snapshot -> measure_and_decide;
201
- on park write the `waiting` record (base_sha/candidate_sha/`candidate_ref`/
201
+ on park write the `parked` record (base_sha/candidate_sha/`candidate_ref`/
202
202
  seed/stage + `experiment_job_id` = the afterany set) and end. A terminal
203
203
  wake `drop_snapshot`s `candidate_ref` (the snapshot outlives every park/wake
204
204
  cycle, so the drop is deferred to run end, never mid-cycle). When a revision
@@ -356,8 +356,8 @@ single writer via lease):
356
356
  | field | role |
357
357
  |---|---|
358
358
  | `run_id` `target` `benchmark` `agent_id` `task_title` `issue_number` `pr_url` | identity/topology |
359
- | `climb_job_id` | the transaction's own job — lets the sweep tell KILLED from crashed; must be re-stamped by any path re-entering `implementing` from a new job |
360
- | `experiment_job_id` `wake_job_id` `followup_job_id` | Slurm handles for the dispatched work, its afterany wake, and review servicing |
359
+ | `climb_job_id` | the transaction's own job — lets the sweep tell KILLED from crashed; must be re-stamped by any path re-entering `running` from a new job |
360
+ | `experiment_job_id` `wake_job_id` | Slurm handles for the dispatched work and its afterany wake |
361
361
  | `resume_session_id` | the harness session a wake reconstructs — the entire "pause" state for a session's mind |
362
362
  | `state` `deadline` `terminal_seen` `wake_attempts` | wake bookkeeping; a waiting record REQUIRES a deadline |
363
363
  | `stage` (phase 1) | a small object, not a label: the parked measure point PLUS the process-local state re-entry needs — `base_sha`, `candidate_sha` and the candidate snapshot's `candidate_ref` (random, so it MUST be stored — a terminal wake hands it to `drop_snapshot` or the snapshot leaks), the drawn `seed` and `suite_seed` (both random, both stored so the wake re-measures PAIRED), the pre-eval tree fingerprints the drift check compares (today locals; a resumed process without them would fail the drift check closed on every dispatch), and the expected result-file names. The scope/suite `measured_paths` are NOT in here — they are re-derived from the `base_sha..candidate_sha` diff (step 4), not stored |
@@ -87,11 +87,9 @@ survival the other. No strawman reimplementation.
87
87
  Drop the "human keeps the merge button" framing as a principle — it is the
88
88
  DEFAULT, not a law. Merge policy becomes a contract knob (`merge:
89
89
  manual | auto`, target owner's declaration, like a harness permission mode):
90
- `auto` = gate + panel clean → the PR merges itself. NOTE the current
91
- arming path is review-required-shaped (`arm_auto_merge_when_review_required`
92
- — GitHub only arms against a pending requirement), so `auto` mode needs a
93
- small publish change: merge directly when the gate CI is the sole
94
- requirement, arm otherwise. Graded
90
+ `auto` = the kernel sweep merges a clean PR at the head approved by the gate
91
+ and panel, with the API's expected-head guard. Publish never arms GitHub
92
+ auto-merge in this mode. Manual mode keeps the required-review guard. Graded
95
93
  autonomy is paper material: interventions per accepted step, measured at
96
94
  each grade.
97
95
 
@@ -0,0 +1,338 @@
1
+ # The run lifecycle: states, messages, and what the kernel decides
2
+
3
+ **Status: lifecycle redesign landed (2026-09-12).** A re-read of the
4
+ whole lifecycle against one rule, from Mengye: benchmark verification and
5
+ launching jobs are rigid; everything else is the author deciding, through
6
+ tool calls. It absorbs Phase C of `research-loop-buildout.md` and supersedes
7
+ the follow-up sections of `roles.md` and `orchestrator-verify.md`. Built from
8
+ three inventories of the code as of main `8944625` (states and transitions,
9
+ inbound messages and syscalls, standing design rulings) and cross-checked by
10
+ an independent read of the note against the code; the numbers in brackets
11
+ refer to the kernel-decision inventory posted on the PR that carries this
12
+ note.
13
+
14
+ ## The rule
15
+
16
+ The kernel owns two things because nobody else can be trusted with them:
17
+
18
+ - **The gate.** The credited number is the kernel's paired measurement of a
19
+ sealed tree on the fixed benchmark under a private seed, with scope checked
20
+ on the diff and the suite re-measured when shared paths moved. The author
21
+ never sees the seed, never edits the ledger, and cannot make a number.
22
+ - **Launching.** Jobs run outside the sandbox under the kernel's containment,
23
+ on the lane the contract names, metered in counts and GPU-hours, always
24
+ queued, cancelled when the run ends.
25
+
26
+ Around those sit invariants that are not science and are not the author's to
27
+ change: who may send a message and who may merge, that a human's commit is
28
+ never overwritten, isolation of the session, crash-proof liveness, a report on
29
+ every ending, the line notebook sealed at every terminal. Everything else is
30
+ judgment, and judgment belongs to the author: what to run, what a result
31
+ means, whether to answer a reviewer with an experiment or a sentence, when to
32
+ submit, when to stop. Where the kernel makes such a call today, that call
33
+ goes.
34
+
35
+ ## What exists
36
+
37
+ The lifecycle is now the three-state one: `running`, `parked`, `ended`.
38
+ A PR is recorded in `pr_url`. The sweep polls parked runs for terminal jobs,
39
+ checkpoint deadlines and inbox messages. GitHub comments and base moves enter
40
+ the inbox; a run sleeping on jobs receives them with the job results. A PR
41
+ at rest waits without a deadline or idle wake cost. Every author leg resumes
42
+ through `attempt.run_author_leg`, including the steward's review work.
43
+
44
+ `followup.py` and its job are gone. GitHub collection positions live beside
45
+ the inbox, while the run record carries the delivered sequence. Replies are
46
+ posted by the author leg; only a submit invokes the gate and one publish.
47
+ The six endings and reports remain. Old state names and collection positions
48
+ migrate on read. Each tick logs `legacy follow-up records: N` from raw live
49
+ records; operators must confirm zero across every fleet before deployment.
50
+
51
+ ## The lifecycle
52
+
53
+ ### Three states
54
+
55
+ | State | Meaning | Leaves it |
56
+ | --- | --- | --- |
57
+ | `running` | a session is live in a job | the session ends: it slept (park), or it stopped (end) |
58
+ | `parked` | no session; the run waits for jobs it launched, for messages, or both | a wake (the same session resumes), or a human ends the PR |
59
+ | `ended` | terminal, with a report | never |
60
+
61
+ A PR being open is a fact about a run, recorded in `pr_url`, not a state. A
62
+ parked run with a PR is what `in-review` was. `implementing` is `running`;
63
+ `waiting` and `in-review` are `parked`; `concluding` is deleted. The six
64
+ endings stay as they are: merged, rejected, negative result, budget exhausted,
65
+ aborted, stuck. They are how a human reads the board, and every one still
66
+ produces a report.
67
+
68
+ ### One engine: park and wake
69
+
70
+ A run parks when its session ends with a sleep. A run wakes when the kernel
71
+ has something to deliver: the jobs the session slept on have finished, or a
72
+ message arrived for a run that is not waiting on jobs. A wake resumes the
73
+ same session with everything in its inbox rendered as one data-fenced text.
74
+ That is the whole engine, the same on every backend and in every phase of a
75
+ run: before a PR, with a PR open, after a reviewer writes. The four park
76
+ shapes become one park with a job list that may be empty; the wake paths
77
+ become one.
78
+
79
+ A sleep on no jobs is a checkpoint. It wakes when a message arrives or at the
80
+ checkpoint deadline, whichever comes first, and the wake text says which. A
81
+ parked run with a PR and no jobs waits for messages without a deadline; idle
82
+ waiting spends no wake attempt and cannot end as stuck.
83
+
84
+ ### Messages
85
+
86
+ A message is the unit the kernel delivers. Every message has a source and an
87
+ origin (the login, job name or run behind that source), a qualified thread
88
+ such as `owner/repo#9` (the PR when one exists, else the issue the run claimed),
89
+ a trust level (everything but the kernel's own budget and clock lines is data, never
90
+ instructions), and an arrival time. The tick writes messages into the run's
91
+ inbox; the wake drains it in arrival order.
92
+
93
+ | Message | Source | Written by | Delivered |
94
+ | --- | --- | --- | --- |
95
+ | launch result: exit code, bounded tails, declared artifacts | the author's own job | the sweep, when the job is terminal | at the wake the sleep asked for |
96
+ | gate verdict: the paired numbers, the floor, the suite, or an eval error; the sealed tree, the base and the contract it was measured under | the kernel's measurement of a submitted tree | the sweep, when the gate jobs are terminal | at the wake |
97
+ | panel verdict: blocking and advisory findings, the transcript | judge sessions | the panel run | at the wake |
98
+ | human comment or review, on the run's thread | a person with standing; others as context, never as triggers | the tick's poll | at the next wake |
99
+ | check result: the current PR head, check name, conclusion, URL and bounded log tail | CI, with the app slug or check name as origin | the tick, once per check run and conclusion | failures wake runs sleeping on no jobs; success, neutral and skipped wait as context |
100
+ | head moved: the PR head no longer matches the blessed head | git | the tick, once per new head in auto mode | at the next wake |
101
+ | base moved: the digest of what merged and the siblings' numbers | git, main | the tick, once per new base | at the next wake |
102
+ | budget and clock: counts remaining, the review top-up, walltime | the kernel | the wake itself | leads every wake |
103
+ | PR merged or closed | a human, on GitHub | the tick's poll | ends the run; not delivered |
104
+
105
+ The author's session never sees a raw comment body outside a fence, never
106
+ sees the seed, and never sees a message the kernel did not write into the
107
+ inbox. Advisory findings are delivered like blocking ones; the author decides
108
+ what to do with them. The sibling view is refreshed at every wake, not only
109
+ at start. The inbox keeps a position per GitHub collection, since issue
110
+ comments, reviews and review comments carry independent id sequences, and
111
+ delivers each message once.
112
+
113
+ Human messages do not interrupt a sleep on jobs. They wait in the inbox and
114
+ arrive with the job results. The author who wants to answer sooner takes a
115
+ checkpoint sleep. The kernel stays out of scheduling.
116
+
117
+ The inbox is files: a directory beside the run's record, one file per
118
+ message named by a sequence number, written by temp-and-rename, never
119
+ deleted, outside the workspace so the session cannot forge one. The record
120
+ holds the delivered position. A message file is durable before any poll
121
+ position advances, and the delivered position advances only after the wake's
122
+ session leg ends, so a wake that dies re-delivers; double delivery is
123
+ harmless, as it is for today's leases. Records, leases and the inbox all sit behind
124
+ one small storage interface (put, list, get, conditional put) with a
125
+ filesystem implementation for Slurm and local compute; a cloud backend gives
126
+ it an object-store implementation and nothing else changes. No messaging
127
+ service is needed: delivery only happens at a wake, which the kernel
128
+ triggers, so nothing waits on a push. A job-completion callback may later
129
+ wake the tick sooner than its cadence; that is a trigger, not a transport.
130
+
131
+ ### The author's moves
132
+
133
+ | Move | What the kernel does | Rigid part |
134
+ | --- | --- | --- |
135
+ | `launch` | stages a contained job for the contract's lane; the author keeps working | metering |
136
+ | `sleep` | seals the tree, submits the staged jobs, records them in the ledger, parks the run on them (possibly none); the results arrive at the wake | containment, placement, the sleep count |
137
+ | `submit` | seals the tree, runs the gate and the panel as jobs, delivers the verdict as a message; a credited verdict publishes | the gate; the publish |
138
+ | `reply` | posts text on the thread the message came from, with secrets redacted and self-approval scrubbed, through the author leg | standing of the poster; redaction |
139
+ | `end` | without a PR, ends with the report and last failed verdict; with an open PR, posts a supplied report and parks for messages | the report |
140
+
141
+ `reply` and `end` are new. A session that stops without sleeping or
142
+ submitting ends the run with what it has, unmeasured. That is a change: today
143
+ a plain finish is measured by the gate on the tree it left, and can publish.
144
+ Under this note only a submit is measured, on every benchmark, so the author
145
+ always asks for its number. The read-only verbs stay: `status`, `queue`,
146
+ `history`, `siblings`, `reports`, `sync`. The steward has the same moves under
147
+ its own key, with its own scope.
148
+
149
+ The research advice now written into the wake text and the brief moves to
150
+ the role's instructions, where it can be edited without a kernel change. The
151
+ kernel's wake text states facts: the messages, the counts, the clock.
152
+
153
+ ### Submit, and the one publish
154
+
155
+ A submit means: measure this sealed tree, and if it is credited, publish it.
156
+ The verdict comes back as a message whatever it says, naming the sealed
157
+ tree, the base and the contract it was measured under. A gate that says no is
158
+ not a terminal; the author reads the numbers and decides to try again, launch
159
+ more, or end. Today a plain finish's failed gate ends the run and a submitted
160
+ park's failed gate wakes the author; the second is right, and with `end` the
161
+ author can still choose the first.
162
+
163
+ The publish is the one rigid act after the gate, and there is one of it. When
164
+ no PR exists, the kernel opens one from the sealed tree with the number and
165
+ the author's report. When a PR exists, the kernel moves its head to the
166
+ sealed tree and posts the number, after confirming auto-merge is disarmed.
167
+ The move is a fast-forward only: the sealed tree must contain the PR's
168
+ current head. When a person pushed to the PR since the author last saw it,
169
+ the publish is refused and the fact is delivered as a message; the author
170
+ merges and submits again. A human's commit is never overwritten. The publish
171
+ is also refused when the base's contract no longer defines the benchmark
172
+ with the measurement the verdict was made under; a stale verdict is said,
173
+ not published.
174
+
175
+ Blocking findings at a publish open a draft, or keep the PR as it is, and are
176
+ delivered to the author with the verdict. The ledger row moves only when a
177
+ credited number beats the recorded best by the floor; every other number is
178
+ posted and leaves the row alone. A submit whose number is worse than the PR's
179
+ current one still moves the head, with the number stated plainly; the author
180
+ chose it, the thread shows it, and the PR is merged or not.
181
+
182
+ A steward submit is a ruler change and is measured as one: the full suite
183
+ runs, every sibling's eval is smoke-checked, and the credited number resets
184
+ the baseline instead of competing with it. That is the steward's publish; the
185
+ steward's copy of the review pipeline is not.
186
+
187
+ ### The meter
188
+
189
+ One meter for the run's whole life: the launch count, the sleep count and the
190
+ GPU-hours the contract sets, spent from the first session to the last. When a
191
+ PR opens the counts get a small top-up the contract names, so review has
192
+ headroom, and every wake states the counts remaining and that the top-up was
193
+ added, so the author can plan for review while still climbing. A reply costs
194
+ nothing. A run that has spent everything can still reply and end.
195
+
196
+ ### Endings
197
+
198
+ A run ends when the author ends it, when its PR is merged or closed,
199
+ when the meter runs out, or when the kernel cannot continue: a crash, a
200
+ tampered workspace, or a kernel action that made no progress
201
+ `MAX_WAKE_ATTEMPTS` times, a failed publish retry included. A merge or close
202
+ ends the run at the next tick whatever it is doing: pending jobs are
203
+ cancelled, a session in flight finishes its leg and its publish is refused.
204
+ Every ending writes the report, seals the line notebook, releases the issue
205
+ claim when no PR exists, and cancels the run's live launches. No other path
206
+ ends a run. A gate verdict never ends a run by itself, and a reviewer's
207
+ comment never does.
208
+
209
+ ## What stays rigid
210
+
211
+ | Kept | Why it is the kernel's |
212
+ | --- | --- |
213
+ | the paired measurement, the private seed, the floor, the suite, the cached baseline rule, the zero-change rule (an unchanged tree cannot be credited), the verdict bound to its sealed tree, base and contract | the number must be nobody's claim |
214
+ | scope on the diff before anything is sealed, launched or measured | the out-of-scope edit could be to the ruler |
215
+ | containment, the lane from the contract, `--nice` on launches, always queue, cancel on end | the session cannot hold GPUs or credentials |
216
+ | launch, sleep and GPU-hour counts; refusal on exhaustion with the numbers | the meter is the only bound on spend |
217
+ | the publish: open or fast-forward the PR head to the sealed tree, the ledger row rule, disarm before a head moves, refuse when a human pushed or the contract moved; under `merge: auto`, record the blessed head and let the sweep merge a clean, quiet PR at that head only, never arm GitHub auto-merge; otherwise humans merge | credit, merge authority, and nobody's work overwritten |
218
+ | the steward's ruler measurement: full suite, sibling smoke checks, baseline reset | a ruler change must be verified as one |
219
+ | standing: which comments are messages, the bot's own markers, the task label; the issue claim and its release; one delivery per message | authorization and liveness |
220
+ | leases, the sweep, deadline floors, the stuck cap, the outage latch, the tamper guard, the report on every ending, the line seal at every terminal | liveness and audit |
221
+
222
+ ## What becomes the author's, and what is deleted
223
+
224
+ | Today | After |
225
+ | --- | --- |
226
+ | an in-scope edit during review is re-measured as the PR candidate [25, 29] | nothing is measured unless submitted |
227
+ | the base-sync ladder: sync, conflict, behind, withheld, superseded, "the numbers above still stand" [25, 26, 30, 31] | one message, base moved; the author merges or not and submits or not |
228
+ | six withhold wordings for a reverted change [29] | the change is never reverted; the author's tree is the author's |
229
+ | abandon a finished re-measure when the head moved [34] | the publish is a fast-forward or a refusal delivered as a message |
230
+ | the panel re-read after a push, its two-revision cap [36, 37] | a submit runs the panel; findings are a message; the sleep count bounds revisions |
231
+ | `finish_attempts` | folds into `wake_attempts`: any kernel retry without progress counts |
232
+ | the kernel re-pins the base and tells the agent to merge; kernel-written conflict prompts [26, 54] | the base-moved message says what happened; no instruction |
233
+ | a submit is refused until the run has launched [49]; a metered finish without a submit is scored no-improvement [7]; a submit needs a report [50] | the meter and the gate are the constraints; the report is `end`'s and the publish's |
234
+ | a plain finish is measured; its failed gate ends the run as negative-result [9] | only a submit is measured; the verdict is a message; `end` is the author's |
235
+ | research advice in the wake text and the brief | the role's instructions |
236
+ | advisory findings never reach the author; siblings frozen at start | delivered; refreshed |
237
+ | two first-publish implementations; an inline submit that drops its sibling launches | one publish; one job-and-wake path on every backend |
238
+ | three cursors as record fields, `followup_stage`, `panel_wake_*`, `dirty_wake_head`, `extra_update`, `improve_prompt` | one inbox with per-collection positions, one renderer |
239
+ | the steward's copy of the pipeline; its mission text in kernel code [43, 45] | the steward is a role with the same moves and its own publish; its text is role configuration |
240
+ | `MAX_COMMENTS_PER_WAKE`, `PANEL_WAKE_CAP`, `finish_attempts`, `panel_wake_rounds` | gone; the sleep count and `wake_attempts` are the bounds |
241
+
242
+ Out of this pass, and named so nobody reads their absence as a decision: the
243
+ outer loop's choices (which issue, which benchmark next, an issue naming one
244
+ benchmark [39–42]) belong to a planner note; the declared-comparison gate and
245
+ the portfolio ledger stay in `research-loop.md`.
246
+
247
+ ## The record
248
+
249
+ ```
250
+ run_id, target, benchmark, agent_id, issue_number
251
+ state: running | parked | ended ending, ending_note
252
+ pr_url the fact a PR is open
253
+ session: backend, model, key_file, resume_session_id
254
+ park: jobs (afterany ids), sealed_ref, base_sha, base_branch, deadline
255
+ inbox: positions per GitHub collection, last delivered message
256
+ meter: launches_used, sleeps_used, gpu_hours_used, review_topup_added
257
+ wake_attempts liveness only
258
+ ```
259
+
260
+ Gone from the record: `followup_stage`, `followup_job_id`, `panel_wake_head`,
261
+ `panel_wake_text`, `panel_wake_rounds`, `dirty_wake_head`. `auto_blessed_head`
262
+ stays only as long as `merge: auto` does.
263
+
264
+ The candidate and submitted phases remain inside `stage`: they describe what
265
+ its jobs measure and what result handling the shared wake must resume. They
266
+ are metadata within the one park, not separate lifecycle states or wake paths.
267
+
268
+ ## Decisions
269
+
270
+ Settled (Mengye, 2026-09-12):
271
+
272
+ 1. A human message waits for the jobs a run sleeps on; no interrupt.
273
+ 2. One meter for the run's life, with a small top-up when the PR opens, stated
274
+ in every wake.
275
+ 3. A worse submit on an open PR moves the head, with the number posted
276
+ plainly; the ledger row does not move.
277
+ 4. `end` is a verb, so a report is asked for at the moment the author
278
+ decides; a session that simply stops still ends the run with what it has.
279
+
280
+ 5. **Auto-merge** (Mengye, 2026-09-13): the owner opts in with `merge: auto`.
281
+ Publish records `auto_blessed_head` and never arms GitHub auto-merge in
282
+ this mode. The sweep merges a clean, quiet PR at that head only, with the
283
+ API's expected-head guard, while its base contract still permits auto.
284
+ Old arms are withdrawn; a changed head is a message to the author.
285
+ Manual mode keeps its required-human-review arming guard.
286
+
287
+ ### Later
288
+
289
+ Multi-agent collaboration and sub-agent teams will need real addressing:
290
+ sender and recipient, with routable messages. That is a future design item.
291
+ The per-run inbox and outbox stay until then. Replies store their qualified
292
+ thread when staged, so a later PR change does not change their destination.
293
+
294
+ ## Sequencing
295
+
296
+ Each stage is one PR, reviewed, run from its commit on one fleet before any
297
+ release, and deletes what it replaces. A fleet never lacks a working path
298
+ between stages.
299
+
300
+ 1. **One inbox, one renderer.** Every inbound message, including the ones
301
+ delivered today, is written to the inbox and rendered by one function;
302
+ pending blocking findings are read from the inbox from this stage on, so
303
+ `panel_wake_text` can go. Advisory findings and a refreshed sibling view
304
+ ride along. No lifecycle change yet.
305
+ 2. **Messages reach a parked author.** Status: landed. `reply` exists. A run with an open PR
306
+ parks; comments and base moves are inbox messages; the author can launch,
307
+ reply and sleep in review. A session that edits code in review still goes
308
+ through today's re-measure-and-push until the next stage replaces it.
309
+ 3. **Submit in review, and `end`.** Status: landed in full, including `end`
310
+ and the review top-up. The publish moves the PR head by
311
+ fast-forward; a gate verdict is a message on every path; the meter's
312
+ top-up exists; the submit policies [7, 49, 50] go; the follow-up
313
+ re-measure path no longer runs.
314
+ 4. **Three states.** Status: landed. `running`, `parked`, `ended`; the record migrates on
315
+ read. This stage lands only when no live record on any fleet carries
316
+ `followup_stage` or a follow-up job; the tick reports the count until it is zero.
317
+ `followup.py`, the steward's pipeline copy, and the counters are deleted
318
+ with their tests.
319
+
320
+ 5. **Session memory, message identities, CI results and direct merging.** Status: landed.
321
+ Memory guidance covers every session boundary. Messages carry origins and
322
+ qualified threads; staged replies keep their destinations. CI results and
323
+ head moves use the inbox. Auto mode merges through the sweep at the blessed
324
+ head and never arms GitHub auto-merge.
325
+
326
+ ## Standing contradictions this settles
327
+
328
+ - Launches are submitted at sleep time, not when `launch` is called: kept.
329
+ Staging is the primitive; the author keeps working after a launch and the
330
+ jobs start when it sleeps.
331
+ - There is a live channel to a running session, the watcher, for read-only
332
+ questions. `roles.md`'s "no live channel" is stale.
333
+ - A follow-up's change is not re-measured by the orchestrator; only a submit
334
+ is measured. `roles.md` and `orchestrator-verify.md` are superseded here.
335
+ - The panel re-read loop after a push is retired; a submit runs the panel.
336
+ - A moved base is told, never forced; the finish's "merge and re-measure
337
+ both sides" in `roles.md`'s flow is retired.
338
+ - One batch in flight per run is not a rule; launch several, sleep once.