outerloop-science 0.2.1__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (279) hide show
  1. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/CHANGELOG.md +271 -0
  2. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/CITATION.cff +2 -2
  3. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/CLAUDE.md +5 -0
  4. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/PKG-INFO +1 -1
  5. outerloop_science-0.3.0/docs/design/accelerators.md +205 -0
  6. outerloop_science-0.3.0/docs/design/agent-environment.md +95 -0
  7. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/agent-protocols.md +1 -1
  8. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/architecture.md +4 -3
  9. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/lifecycle.md +48 -11
  10. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/research-loop-buildout.md +3 -2
  11. outerloop_science-0.3.0/docs/endpoints.md +252 -0
  12. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/install.md +358 -6
  13. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/pyproject.toml +1 -0
  14. outerloop_science-0.3.0/scripts/install_bridge.sh +33 -0
  15. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/install_claude.sh +26 -12
  16. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/install_codex.sh +40 -11
  17. outerloop_science-0.3.0/scripts/install_hermes.sh +109 -0
  18. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/tick_chain.sbatch +15 -1
  19. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/tick_deploy.sh +81 -73
  20. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/tick_resident.sh +5 -0
  21. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/__init__.py +1 -1
  22. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/attempt.py +649 -134
  23. outerloop_science-0.3.0/src/outerloop/author_overrides.py +147 -0
  24. outerloop_science-0.3.0/src/outerloop/bridge_install.py +35 -0
  25. outerloop_science-0.3.0/src/outerloop/bridge_runtime/pyproject.toml +5 -0
  26. outerloop_science-0.3.0/src/outerloop/bridge_runtime/uv.lock +1811 -0
  27. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/brief.py +15 -6
  28. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/cli.py +218 -26
  29. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/climbboard.py +34 -10
  30. outerloop_science-0.3.0/src/outerloop/codex_bridge.py +529 -0
  31. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/compute.py +200 -9
  32. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/dispatch.py +36 -2
  33. outerloop_science-0.3.0/src/outerloop/endpoint_wait.py +132 -0
  34. outerloop_science-0.3.0/src/outerloop/endpoints.py +261 -0
  35. outerloop_science-0.3.0/src/outerloop/gpu_lanes.py +92 -0
  36. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/harness.py +219 -42
  37. outerloop_science-0.3.0/src/outerloop/harness_cli.py +401 -0
  38. outerloop_science-0.3.0/src/outerloop/harness_pins.py +75 -0
  39. outerloop_science-0.3.0/src/outerloop/harnesses.toml +17 -0
  40. outerloop_science-0.3.0/src/outerloop/hermes_install.py +34 -0
  41. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/init.py +185 -29
  42. outerloop_science-0.3.0/src/outerloop/instance.py +30 -0
  43. outerloop_science-0.3.0/src/outerloop/job_names.py +23 -0
  44. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/launchlog.py +14 -1
  45. outerloop_science-0.3.0/src/outerloop/limits.py +151 -0
  46. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/measure.py +69 -13
  47. outerloop_science-0.3.0/src/outerloop/operator_limits.py +220 -0
  48. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/orchestrator.py +284 -78
  49. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/panel.py +14 -2
  50. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/paths.py +15 -0
  51. outerloop_science-0.3.0/src/outerloop/provenance.py +33 -0
  52. outerloop_science-0.3.0/src/outerloop/rebind.py +229 -0
  53. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/review.py +7 -1
  54. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/review_agent.py +23 -6
  55. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/review_agent_cli.py +20 -0
  56. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/role_runner.py +40 -6
  57. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/runstate.py +63 -6
  58. outerloop_science-0.3.0/src/outerloop/status.py +115 -0
  59. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/tick.py +466 -105
  60. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/verifier.py +14 -6
  61. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/watcher.py +9 -1
  62. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/conftest.py +16 -0
  63. outerloop_science-0.3.0/tests/fixtures/README.md +80 -0
  64. outerloop_science-0.3.0/tests/fixtures/author_route_legacy.json +31 -0
  65. outerloop_science-0.3.0/tests/fixtures/author_route_missing_model.json +30 -0
  66. outerloop_science-0.3.0/tests/fixtures/author_route_missing_route.json +28 -0
  67. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
  68. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/exit-code +1 -0
  69. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
  70. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/provenance.json +1 -0
  71. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/stdout +1 -0
  72. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
  73. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/identity.json +4 -0
  74. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
  75. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
  76. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/provenance.json +1 -0
  77. outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
  78. outerloop_science-0.3.0/tests/fixtures/hermes_resume_legacy.json +1 -0
  79. outerloop_science-0.3.0/tests/fixtures/hermes_sample_20260924.json +28 -0
  80. outerloop_science-0.3.0/tests/fixtures/launches-before-provenance.jsonl +2 -0
  81. outerloop_science-0.3.0/tests/fixtures/rc1_v021/baselines/main@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.json +1 -0
  82. outerloop_science-0.3.0/tests/fixtures/rc1_v021/brief.txt +44 -0
  83. outerloop_science-0.3.0/tests/fixtures/rc1_v021/end-request.json +1 -0
  84. outerloop_science-0.3.0/tests/fixtures/rc1_v021/ended.json +44 -0
  85. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-provenance/command.txt +1 -0
  86. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-provenance/job.sh +21 -0
  87. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
  88. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/exit-code +1 -0
  89. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
  90. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/stdout +1 -0
  91. outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
  92. outerloop_science-0.3.0/tests/fixtures/rc1_v021/generate.py.txt +72 -0
  93. outerloop_science-0.3.0/tests/fixtures/rc1_v021/identity.json +1 -0
  94. outerloop_science-0.3.0/tests/fixtures/rc1_v021/launches.jsonl +1 -0
  95. outerloop_science-0.3.0/tests/fixtures/rc1_v021/line.bundle +0 -0
  96. outerloop_science-0.3.0/tests/fixtures/rc1_v021/merged.json +44 -0
  97. outerloop_science-0.3.0/tests/fixtures/rc1_v021/open.json +44 -0
  98. outerloop_science-0.3.0/tests/fixtures/rc1_v021/pending/owner__repo.json +1 -0
  99. outerloop_science-0.3.0/tests/fixtures/rc1_v021/pending/owner__repo@agent-01.json +1 -0
  100. outerloop_science-0.3.0/tests/fixtures/rc1_v021/pr-states.json +1 -0
  101. outerloop_science-0.3.0/tests/fixtures/rc1_v021/queue.json +1 -0
  102. outerloop_science-0.3.0/tests/fixtures/rc1_v021/rebind.json +1 -0
  103. outerloop_science-0.3.0/tests/fixtures/rc1_v021/sessionless.json +44 -0
  104. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_attempt.py +1110 -24
  105. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_attempt_review.py +58 -32
  106. outerloop_science-0.3.0/tests/test_author_overrides.py +777 -0
  107. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_brief.py +1 -1
  108. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_climbboard.py +59 -1
  109. outerloop_science-0.3.0/tests/test_codex_bridge.py +831 -0
  110. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_compute.py +119 -0
  111. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_default_claude_model.py +12 -1
  112. outerloop_science-0.3.0/tests/test_endpoint_wait.py +256 -0
  113. outerloop_science-0.3.0/tests/test_endpoints.py +849 -0
  114. outerloop_science-0.3.0/tests/test_gpu_lanes.py +243 -0
  115. outerloop_science-0.3.0/tests/test_harness_pins.py +600 -0
  116. outerloop_science-0.3.0/tests/test_hermes_author.py +195 -0
  117. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_hermes_harness.py +164 -12
  118. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_inbox.py +46 -0
  119. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_init.py +166 -1
  120. outerloop_science-0.3.0/tests/test_install_harness.py +557 -0
  121. outerloop_science-0.3.0/tests/test_instance.py +96 -0
  122. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_launchlog.py +25 -0
  123. outerloop_science-0.3.0/tests/test_limits.py +156 -0
  124. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_local_compute.py +26 -0
  125. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_measure.py +107 -0
  126. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_measure_and_decide.py +93 -0
  127. outerloop_science-0.3.0/tests/test_operator_end.py +350 -0
  128. outerloop_science-0.3.0/tests/test_operator_limits.py +653 -0
  129. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_orchestrator.py +910 -9
  130. outerloop_science-0.3.0/tests/test_paths.py +76 -0
  131. outerloop_science-0.3.0/tests/test_rebind.py +774 -0
  132. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review.py +2 -0
  133. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review_agent.py +69 -0
  134. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_runstate.py +13 -0
  135. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_start.py +74 -7
  136. outerloop_science-0.3.0/tests/test_status.py +123 -0
  137. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_tick.py +147 -4
  138. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_tick_resident.py +176 -3
  139. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_verifier.py +7 -3
  140. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_version.py +1 -1
  141. outerloop_science-0.2.1/scripts/install_hermes.sh +0 -43
  142. outerloop_science-0.2.1/src/outerloop/limits.py +0 -80
  143. outerloop_science-0.2.1/tests/test_install_harness.py +0 -218
  144. outerloop_science-0.2.1/tests/test_limits.py +0 -71
  145. outerloop_science-0.2.1/tests/test_paths.py +0 -23
  146. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/.gitignore +0 -0
  147. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/.pre-commit-config.yaml +0 -0
  148. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/.python-version +0 -0
  149. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/CONTRIBUTING.md +0 -0
  150. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/LICENSE +0 -0
  151. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/NOTICE +0 -0
  152. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/README.md +0 -0
  153. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/RELEASING.md +0 -0
  154. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/SECURITY.md +0 -0
  155. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/containers/README.md +0 -0
  156. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/containers/agent-py312.def +0 -0
  157. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/assets/icon-dark.svg +0 -0
  158. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/assets/icon-light.svg +0 -0
  159. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/assets/icon.svg +0 -0
  160. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/community.md +0 -0
  161. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/compute.md +0 -0
  162. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/contract.md +0 -0
  163. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/agent-substrate.md +0 -0
  164. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/base-reintegration.md +0 -0
  165. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/consolidation.md +0 -0
  166. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/dispatcher.md +0 -0
  167. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/eval-cache.md +0 -0
  168. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/external.md +0 -0
  169. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/github-app-auth.md +0 -0
  170. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/headline.md +0 -0
  171. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/judge-placement.md +0 -0
  172. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/meta.md +0 -0
  173. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/multi-cluster.md +0 -0
  174. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/onboarding.md +0 -0
  175. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/orchestrator-verify.md +0 -0
  176. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/public-surface.md +0 -0
  177. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/research-lines.md +0 -0
  178. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/research-loop.md +0 -0
  179. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/resident-tick.md +0 -0
  180. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/review-placement.md +0 -0
  181. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/reviewer-infra.md +0 -0
  182. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/role-cli.md +0 -0
  183. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/roles.md +0 -0
  184. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/scaling.md +0 -0
  185. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/design/session-watcher.md +0 -0
  186. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/reviewer.md +0 -0
  187. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/roadmap.md +0 -0
  188. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/docs/validation/author-syscalls.md +0 -0
  189. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/examples/review.yml +0 -0
  190. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/README.md +0 -0
  191. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/requeue_moved_successors.sh +0 -0
  192. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/setup_branch_protection.sh +0 -0
  193. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/scripts/sweep_git_locks.sh +0 -0
  194. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/__main__.py +0 -0
  195. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/appauth.py +0 -0
  196. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/appmanifest.py +0 -0
  197. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/contract.py +0 -0
  198. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/contract_cli.py +0 -0
  199. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/disk.py +0 -0
  200. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/evalcache.py +0 -0
  201. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/github.py +0 -0
  202. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/housekeeping.py +0 -0
  203. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/hypothesis.py +0 -0
  204. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/image.py +0 -0
  205. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/inbox.py +0 -0
  206. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/intake.py +0 -0
  207. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/ledger_branch.py +0 -0
  208. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/ledger_events.py +0 -0
  209. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/ledger_migrate.py +0 -0
  210. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/maintain.py +0 -0
  211. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/maintain_agent_cli.py +0 -0
  212. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/maintain_post_cli.py +0 -0
  213. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/markers.py +0 -0
  214. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/posting.py +0 -0
  215. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/progress.py +0 -0
  216. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/py.typed +0 -0
  217. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/review_post_cli.py +0 -0
  218. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/review_summarize_cli.py +0 -0
  219. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/roles.py +0 -0
  220. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/rolespec.py +0 -0
  221. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/steward.py +0 -0
  222. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/style.py +0 -0
  223. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/syscall.py +0 -0
  224. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/syscall_cli.py +0 -0
  225. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/verify_agent.py +0 -0
  226. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/verify_agent_cli.py +0 -0
  227. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/src/outerloop/verify_post_cli.py +0 -0
  228. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/fakes.py +0 -0
  229. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/fixtures/base_moved_cc4e5d7.json +0 -0
  230. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/fixtures/panel_wake_9ca3d7c.json +0 -0
  231. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/helpers.py +0 -0
  232. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/ledger_fake.py +0 -0
  233. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_app_permissions.py +0 -0
  234. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_appauth.py +0 -0
  235. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_appmanifest.py +0 -0
  236. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_bot_aliases.py +0 -0
  237. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_bot_login.py +0 -0
  238. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_channel_dir.py +0 -0
  239. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_codex_harness.py +0 -0
  240. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_contract.py +0 -0
  241. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_contract_cli.py +0 -0
  242. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_contract_names.py +0 -0
  243. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_disk.py +0 -0
  244. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_dispatch.py +0 -0
  245. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_evalcache.py +0 -0
  246. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_github.py +0 -0
  247. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_hardening.py +0 -0
  248. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_harness.py +0 -0
  249. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_housekeeping.py +0 -0
  250. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_image.py +0 -0
  251. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_import.py +0 -0
  252. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_intake.py +0 -0
  253. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_ledger_events.py +0 -0
  254. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_ledger_migrate.py +0 -0
  255. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_lifecycle_cli.py +0 -0
  256. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_maintain.py +0 -0
  257. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_markers.py +0 -0
  258. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_messages.py +0 -0
  259. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_packaging.py +0 -0
  260. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_panel.py +0 -0
  261. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_posting.py +0 -0
  262. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_progress.py +0 -0
  263. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_requeue_moved_successors.py +0 -0
  264. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review_agent_cli.py +0 -0
  265. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review_hardening.py +0 -0
  266. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review_policy.py +0 -0
  267. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_review_summarize.py +0 -0
  268. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_role_runner.py +0 -0
  269. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_rolespec.py +0 -0
  270. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_steward.py +0 -0
  271. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_sweep_git_locks.py +0 -0
  272. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_syscall.py +0 -0
  273. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_syscall_cli.py +0 -0
  274. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_tick_chain_successors.py +0 -0
  275. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_tiers.py +0 -0
  276. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_verify_agent.py +0 -0
  277. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_verify_agent_cli.py +0 -0
  278. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/tests/test_watcher.py +0 -0
  279. {outerloop_science-0.2.1 → outerloop_science-0.3.0}/uv.lock +0 -0
@@ -6,6 +6,276 @@ Versions follow [SemVer](https://semver.org).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.3.0] - 2026-10-05
10
+
11
+ ### Upgrading
12
+
13
+ Operator actions (everything else needs no action; details in each entry):
14
+
15
+ - Claude Code is pinned at 2.1.285: run `outerloop harness upgrade claude`
16
+ (deploys that run harness upgrades do this on their own).
17
+ - Before selecting a chat-only Codex endpoint profile, install the bridge:
18
+ `outerloop harness upgrade --used`.
19
+ - Existing Hermes source-only installs: run `bash scripts/install_hermes.sh
20
+ "$REVIEW_HERMES_REPO"` (or full `outerloop init`) before Hermes sessions launch.
21
+ - Harness version overrides now need matching SHA-256 settings.
22
+ - Evals and baselines recorded by the previous kernel are measured again once
23
+ under the new cache key, with no extra charge.
24
+ - Parked authors keep the instructions they started with; judges use the new
25
+ rubric at once.
26
+ - Runs parked by the previous kernel resume after the upgrade.
27
+ - Before rolling back: consume pending rebind requests, finish capacity-parked
28
+ runs, runs with extended session limits, overridden and endpoint-routed runs,
29
+ and chat-only Codex sessions, and stop additional instances.
30
+
31
+ - A PR tests one idea. It may include the few changes that idea needs, and the
32
+ report states the effect of each change. Authors test several values in one
33
+ array launch and report the results around the chosen value; the panel may
34
+ block a single-value tuning change without them (new `landscape` finding
35
+ category). Upgrading: no action. Judges use the new rubric at once. An author
36
+ parked across the upgrade keeps the instructions it started with. Older readers
37
+ treat `landscape` as `other`. (This was listed under 0.2.1 by mistake; it
38
+ shipped after that tag.)
39
+
40
+ - Include GPU count and resolved GPU type in dispatched eval and baseline cache identity, including budget discounts. Legacy eval slots and baseline entries are cache misses. Upgrading: no action. An eval or baseline recorded by the previous kernel, in flight or finished, is measured again once under the new cache key, with no extra budget charge; pausing or draining before the upgrade does not avoid it.
41
+
42
+ - Park launch capacity refusals in capacity wait when an immediate resume is
43
+ unavailable or already refused, instead of ending the run. The next wake
44
+ delivers the refusal and a retry note, resuming the same session when supported
45
+ or starting a fresh session with the run context otherwise. Meters are
46
+ preserved and capacity waits do not exhaust stuck retries.
47
+ Upgrading: no action; existing records need no migration. A run parked by the
48
+ previous kernel in author-sleep with no session and no pending job starts a
49
+ fresh author leg on its next wake instead of ending as a session error.
50
+ Finish capacity-parked runs before rolling back; older kernels cannot resume
51
+ them.
52
+
53
+ - Add operator `outerloop rebind <run-id> [--root <root>] [--note <text>]` to
54
+ explicitly move an existing run to its slot's current author at the next leg,
55
+ including runs waiting on retired endpoints. Reports and board details show
56
+ ordered authors; evaluation and launch provenance retain the producing author.
57
+ Pending requests can be cancelled; three failed applications stop retries,
58
+ with the count and last error visible in status. A new request replaces a failed one.
59
+ - Upgrading: optional run fields `author_history`, `author_rebind_id` and
60
+ `stage.candidate_author` / `stage.candidate_authors`, plus `rebind.json` requests
61
+ and evaluation provenance, require no migration. Legacy author history is populated lazily. Consume
62
+ pending requests before rolling back to a kernel without rebind support.
63
+
64
+ - Add `outerloop end <run-id> [--root <root>] [--note <text>]` to request an
65
+ operator ending at the next tick, including runs waiting for an unavailable
66
+ endpoint. The tick uses the existing ending cleanup and refuses later publish.
67
+ - Upgrading: existing run records need no backfill; an absent `end-request.json`
68
+ means no request. Older kernels ignore requests and do not recognize the new
69
+ `operator` ending when writing records; keep the updated kernel for these runs.
70
+
71
+ - Author overrides accept operator-only `session_minutes` (10–240) and
72
+ `session_max_turns` (10–300). Limits bind with the author selection and survive
73
+ settings changes across wakes and review replies. Contracts can still lower
74
+ budgets; job walltime follows session duration and the operator job cap.
75
+ Judge budgets are unchanged.
76
+ - Upgrading: the new override fields are optional; existing settings and legacy
77
+ run records need no migration. Finish runs using extended limits before
78
+ rolling back to a version without bound author limits.
79
+
80
+ - Bad author overrides hold only affected fresh claims during ticks, with one log
81
+ per entry per tick; malformed settings hold named targets, or all fresh claims
82
+ when unreadable. Other tick services and bound runs continue; `outerloop start`
83
+ and `outerloop init` remain strict. No persisted state changes.
84
+
85
+ - Add read-only `outerloop status` (text/`--json`) for local runs and endpoint
86
+ outages. Endpoint waits stay out of the published board/status strip and never
87
+ trigger research-log commits; log one shared outage start and recovery with
88
+ duration and run IDs.
89
+ Validate optional served model and expiry in bounded endpoint address records.
90
+ - Upgrading: no action needed; the first endpoint deferral adds
91
+ `stage.endpoint_wait` and an `endpoint-waits/<profile>.json` log latch. Missing
92
+ keys/journals are tolerated; ended runs and in-flight PRs are unchanged.
93
+ Rollback to the preceding kernel safely ignores the additive state.
94
+
95
+ - Startup validation of `OUTERLOOP_AUTHOR_OVERRIDES` (`outerloop start`) uses the
96
+ image sessions actually run with, the default image when `OUTERLOOP_IMAGE` is unset. Before, a
97
+ codex override on a deployment without `OUTERLOOP_IMAGE` failed validation and stopped the tick.
98
+
99
+ - `OUTERLOOP_AUTHOR_OVERRIDES` accepts a list of entries per target, so different
100
+ slots of one target can use different authors (each listed entry names its slots;
101
+ a slot may appear only once). The single-object form is unchanged.
102
+
103
+ - Reject Codex authors and judges on chat-only endpoints during preflight when
104
+ the bridge runtime is missing or stale, with `outerloop harness upgrade --used`
105
+ as the fix, before spending the author budget.
106
+
107
+ - Codex authors and judges can use chat-only endpoint profiles through a
108
+ per-session LiteLLM bridge and streaming shim, including tool calls and
109
+ reasoning replay on resume. The bridge runtime is isolated, transitively
110
+ locked, installed by `outerloop harness upgrade`, and mounted read-only.
111
+ Direct Responses profiles keep their existing route and interruption behavior;
112
+ the Codex pin is unchanged. Bridge runtime defaults use `OUTERLOOP_CACHE_ROOT`,
113
+ else `$OUTERLOOP_ROOT/cache` (local fallback: `~/.outerloop/cache`). Harness
114
+ upgrades skip bridge detection for misconfigured roles with a diagnostic and
115
+ continue upgrading the other configured harnesses.
116
+ - Upgrading: install the bridge with `outerloop harness upgrade --used` before
117
+ selecting chat-only Codex profiles. Existing run records, session homes, and
118
+ harness retry records retain their formats; no backfill is needed. In-flight
119
+ direct-endpoint runs are unchanged. Before rollback, finish chat-only Codex
120
+ sessions or select a Responses-capable profile; older kernels reject that route.
121
+
122
+ - Research-line salvage and terminal snapshots now recheck scope admission
123
+ against the trusted contract before sealing. Out-of-scope changes are
124
+ dropped from the seal (tracked paths retain their parent content), while
125
+ admitted work and line memory survive normal endings and crashes after
126
+ scope refusals. Filtering leaves working files and the real index untouched
127
+ and logs a bounded list of dropped paths. Upgrading: no action; existing line
128
+ branches are left as they are, and the next snapshot drops out-of-scope paths.
129
+ Rollback is safe.
130
+
131
+ - Support separate instances on one cluster account: process-only absolute
132
+ `OUTERLOOP_ENV_FILE`, with the existing ownership/write-permission checks,
133
+ and stable settings-path suffixes for resident and per-cadence scheduler jobs.
134
+ Init, launch, deploy, harness status, and successor recovery use the selected
135
+ settings and instance identity.
136
+ - Upgrading: no action needed for the default fleet; its settings path and job
137
+ names remain unchanged. Existing run records, leases, and heartbeats need no
138
+ migration. Stop additional instances before rolling back to a version without
139
+ instance isolation.
140
+
141
+ - Launch, submit, and stale-submit checkpoint scope violations now refuse every
142
+ request and resume the author with the offending paths and bounded allowed
143
+ scope in the kernel inbox. Later refusals say “Refused again:”. Refusals
144
+ repeat and never end the run; session walltime and contract sleep, launch,
145
+ and GPU-hour budgets bound the loop. Refusal seals nothing, runs no jobs or
146
+ measurements, and spends no request budget. The authoritative measurement
147
+ scope check remains terminal. Abandoning a refused tree ends normally
148
+ without measurement, a line snapshot, or a push; outage and budget endings
149
+ also preserve the rejection. Scope admission precedes malformed request
150
+ and budget refusals.
151
+ - Upgrading: no action or backfill needed. Scope refusals use the existing
152
+ kernel note payload and refusal keys; old inboxes, ended runs, and in-flight
153
+ runs remain readable without rewriting delivered messages. The next request
154
+ uses the new admission behavior. The rejection flag is in-memory only;
155
+ no run-record fields change; rollback is
156
+ safe and restores terminal scope admission checks.
157
+
158
+ - Deployment author overrides select a backend/model per target or agent slot,
159
+ bind it to each run, and leave panel and CI reviewer inheritance on the fleet
160
+ author. Per-run board details identify the effective author and overrides.
161
+ - Endpoint profiles accept an absolute `URL_FILE` instead of `URL`, reading bare
162
+ URLs or JSON addresses on every session. Missing files, failed bounded health
163
+ checks, and interrupted checks defer sessions without spending wake retries.
164
+ Init provisions fleet and override backends before validating prerequisites.
165
+ - Upgrading: no action or backfill needed; records without `author_overridden`
166
+ retain their existing author and panel behavior, including ended runs. New
167
+ overridden records reuse the saved author route and add this optional flag.
168
+ Rollback reads the records but loses fleet-only panel inheritance for overrides;
169
+ finish overridden runs and fresh endpoint capacity parks before rolling back.
170
+ Fresh endpoint deferrals reuse author-sleep capacity parks without a session ID;
171
+ older kernels cannot resume those parks.
172
+
173
+ - Launch ledger submissions require dispatched launch job IDs, preventing stale
174
+ checkpoints from attributing discarded launches to a commit. Gate capacity
175
+ waits still record any dispatched sibling launches. Existing ledger rows and
176
+ run records remain readable and unchanged; no schema change or backfill.
177
+
178
+ - PR measurement tables identify the base and candidate commits and the shared
179
+ eval command from the base tree; experiment tables identify launch commits.
180
+ Verify and review briefs caution against comparing numbers across commits
181
+ without checking history.
182
+ - Upgrading: no action or backfill needed; the first tick tolerates launch ledger
183
+ rows without `commit` (shown as unknown), including ended runs and in-flight
184
+ PRs. New launch records include the sealed commit; existing records and PR
185
+ bodies are not rewritten. Rollback is safe: older readers ignore the added
186
+ field.
187
+
188
+ - Hermes author resumes that exceed the replay budget stay parked with a
189
+ configuration-blocked status, retaining their session and snapshot without
190
+ consuming wake retries. Author/judge separation checks effective key paths
191
+ and credential values before constructing sessions; init rejects incomplete
192
+ Hermes configuration only for Hermes authors. Endpoint author credentials are
193
+ shared by attempt and tick preflight, including native panel key comparisons.
194
+ - Upgrading: no action or backfill needed; legacy records without
195
+ `stage.hermes_resume_required_chars` are unblocked. The first oversized wake
196
+ records the required budget; raising `OUTERLOOP_HERMES_RESUME_MAX_CHARS` lets
197
+ the next tick or wake resume. Successful resumes and endings clear the marker,
198
+ so later normal sleeps are not configuration wakes. Before rollback, resolve
199
+ blocked runs: older kernels ignore this optional field and may consume retries
200
+ or abort them.
201
+
202
+ - Hermes is an author peer: init, native provider/endpoint validation, contained
203
+ fresh and resumed sessions, absolute syscall commands, and separate author keys.
204
+ Resume replay preserves the original brief and latest results within
205
+ `OUTERLOOP_HERMES_RESUME_MAX_CHARS` (default 120000), with explicit omission counts.
206
+ - Upgrading: no backfill; existing records and full saved transcripts remain
207
+ readable. The first Hermes wake applies the replay bound. Finish Hermes author
208
+ runs before rollback: older kernels reject unsupported Hermes author wakes
209
+ (the endpoint-profile predecessor accepts endpoint routes only). Ended records
210
+ remain readable. See [Hermes setup and compatibility](docs/install.md).
211
+
212
+ - Endpoint profiles declare compatible APIs and use `model[endpoint=profile]`
213
+ selectors, preserving native vendor model IDs. Judge credentials enforce file
214
+ separation; verdicts redact the session key before posting or aggregation.
215
+ - Upgrading: legacy records missing model/route fields retain native routing;
216
+ missing native model configuration fails closed instead of adopting fleet endpoints.
217
+
218
+
219
+ - Named, file-authenticated endpoint profiles work for authors, panel lenses,
220
+ and standalone reviewers on Claude Code, Codex, and Hermes. Sessions use each
221
+ backend's native API configuration; keys reach contained sessions through env,
222
+ never argv. See [endpoint settings and validation](docs/endpoints.md).
223
+ - Hermes is pinned to v2026.9.24 (`f97608f178d1ffeca59860195ab7da295f7c8e5f`).
224
+ Remove Fire quoting for its new argparse entrypoint, enable
225
+ `model.reasoning_echo` for endpoint profiles, and sanitize its instruction-file
226
+ aliases in judge checkouts.
227
+ - Upgrading: existing runs need no backfill; the first tick validates endpoint
228
+ selections without changing legacy routes. New endpoint authors save
229
+ `model[endpoint=profile]` selectors; keep those profiles until runs finish, and finish
230
+ endpoint runs before rolling back. Hermes source/runtime must be upgraded
231
+ with the harness. See [compatibility and rollback](docs/endpoints.md#upgrade-and-rollback).
232
+
233
+ ### Fixed
234
+
235
+ - Manual harness upgrades honor `--root`, environment, and `.env` state roots. Retry records are replaced atomically; unreadable or invalid records are logged and ignored. Deploy loads the configured cache root before selecting the uv cache.
236
+ - Harness installers reinstall changed binaries rather than refusing repair; Codex checks its installed binary digest separately from the archive pin, and Hermes runtime reuse checks the interpreter digest.
237
+
238
+ - Harness upgrades enforce a per-harness deadline, kill timed-out installer process groups, restrict installer environments, and back off failed pins while deleting failed candidates. Overrides require explicit integrity hashes; installed binaries are hash-checked before reuse. Workflow pin resolution fails explicitly on older reviewer refs without a pins reader.
239
+
240
+ - Tick entrypoints export a state-root uv cache before running Python. Fleet job environments preserve explicit cache paths and default per-user caches below `OUTERLOOP_CACHE_ROOT` (or the state root).
241
+
242
+ - Run-owned launch, evaluation, and wake job names stay within 128 characters, retaining a stable run key when shortened. Normal names stay unchanged; GPU usage, queue attribution, evaluation deduplication, and flight retention recognize the bounded names.
243
+ - Intake admission counts queued attempts using the existing pending markers, including jobs queued beyond the marker TTL.
244
+
245
+ - GPU accounting accepts Slurm 25.05 wrapped numeric fields and legacy integers, recognizes typed GPU requests and per-node counts, and limits pending array remainders to available throttle slots.
246
+ - CI Hermes provisioning uses the shared runtime installer with anonymous clone retries and matching workflow pins. A source-specific lock protects checkout and runtime mutations; Python discovery excludes active virtualenvs.
247
+ - Panel preflight checks Hermes runtime readiness and explains installation; full init preserves review model and provider settings with environment precedence.
248
+
249
+ - Contained Hermes sessions can start with read-only source; sessions no longer reinstall dependencies or attempt an editable project build.
250
+
251
+ ### Added
252
+
253
+ - Optional per-target GPU lanes route evals and author launches to deployment-specific partitions, accounts, GPU types, and sbatch flags.
254
+
255
+ - Packaged `harnesses.toml` owns Claude, Codex, and Hermes pins. `outerloop harness status` reports installed versions, paths, drift, and operator overrides; `harness upgrade [name...]` verifies versioned installations before atomically recording their paths. Successful kernel deploys upgrade only configured backends; failures retain the previous installation.
256
+
257
+ - Live, tighten-only `<root>/limits.toml` GPU and active-attempt ceilings, with global defaults and per-target sections. Scheduler-reported GPU usage covers pending and running experiments, sweeps, evaluations, and GPU-bearing sessions. Authors receive uncharged launch refusals; evaluations wait for capacity. Lowering a ceiling does not cancel jobs.
258
+ - Read-only `outerloop limits` reports operator ceilings and fleet-owned running/pending GPU usage.
259
+
260
+ ### Changed
261
+
262
+ - Claude Code pin 2.1.272 -> 2.1.285, the current stable release;
263
+ `claude-opus-5-5` refuses Claude Code older than 2.1.280.
264
+ - Upgrading: legacy Codex archive-only markers and Hermes runtimes without interpreter digests are reinstalled on upgrade; legacy Hermes runtimes remain launchable. Existing retry records remain readable, and corrupt records are treated as empty. Run state and in-flight PRs are unchanged; rollback leaves the additional digest files unused.
265
+
266
+ - Upgrading: version overrides now require matching SHA-256 settings; legacy Codex installs without hash markers are reprovisioned. New retry state and hash markers are ignored by older kernels; the first successfully synced tick verifies configured harnesses and records new paths only when needed. Legacy `.env` paths and Hermes runtimes remain readable; old artifacts are retained. See `docs/install.md` for rollback across kernel pins.
267
+
268
+ - Hermes installs a standalone Python and venv once per pinned commit in a sibling runtime, then launches Python directly. Full `init` provisions configured Hermes judges and records their source path; `--no-install-harness` opts out.
269
+
270
+ ### Upgrading notes for the entries above
271
+
272
+ - No action needed; OUTERLOOP_GPU_LANES is optional.
273
+
274
+ - Upgrading: full run-ID names and legacy 60-character queue names remain readable; shortened names use a derived run key without changing run records. Intake adds `@intake-<issue>` files in the existing pending directory; legacy unsuffixed and agent-slot markers remain readable. Drain queued intake jobs from older submitters (which wrote no marker) before relying on attempt ceilings. Upgrade all kernels together; older kernels do not recognize shortened names or intake markers, so drain those jobs before rollback.
275
+
276
+ - Upgrading: the optional `stage.capacity_wait` flag tolerates missing fields; existing state records need only their target for scheduler attribution. No contract schema change or admission journal. Drain older jobs whose names omit the full run ID (and older local jobs without scheduler metadata), and upgrade all submitters before relying on ceilings. Concurrent admissions may overshoot by one batch for two simultaneous checks; no cross-node admission lock.
277
+ - Upgrading: existing Hermes source-only installs require `bash scripts/install_hermes.sh "$REVIEW_HERMES_REPO"` (or full `outerloop init --force` with Hermes configured) to create the persisted runtime. Run records and resume transcripts are unchanged; rollback leaves the sibling runtime unused.
278
+
9
279
  ## [0.2.1] - 2026-09-25
10
280
 
11
281
  ### Upgrading
@@ -45,6 +315,7 @@ Versions follow [SemVer](https://semver.org).
45
315
 
46
316
  ### Changed
47
317
 
318
+ - An idea with a clear mechanism that does not yet beat the best is reported as a success and kept on the author's research line.
48
319
  - The sweep rechecks ancestry for PRs held only by a base-moved blessing.
49
320
  - Merges performed by the sweep are observed and confirmed in the same tick.
50
321
  - Scope checks compare the full candidate tree with its merge-base against the fetched base tip; changes that landed on the base branch never count as the author's, and author edits to the ledger files are refused like any other out-of-scope path.
@@ -13,5 +13,5 @@ authors:
13
13
  repository-code: "https://github.com/outerloop-science/outerloop"
14
14
  url: "https://outerloop.science"
15
15
  license: Apache-2.0
16
- version: 0.2.1
17
- date-released: "2026-09-25"
16
+ version: 0.3.0
17
+ date-released: "2026-10-05"
@@ -22,6 +22,11 @@ uv run pre-commit run --all-files
22
22
  `.github/` are forbidden write paths everywhere, regardless of contract YAML.
23
23
  - Budget caps are load-bearing safety features, not tunables to raise casually.
24
24
  - Never commit credentials, transcripts, or run artifacts (SECURITY.md).
25
+ - This repository is public: code, tests, docs, commit messages and PR
26
+ descriptions carry no deployment specifics (cluster, account, partition or
27
+ node names, or what hardware a lab has). Use generic examples such as
28
+ `owner/repo`, `my-account`, `gpu-large`, and motivate a change by the general
29
+ need.
25
30
  - Merge commits only; never rebase, squash, or force-push.
26
31
  - **Review until quiet**: development PRs iterate advisory-review rounds
27
32
  (after a fix commit, remove then re-add the `autoresearch:review` label
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: outerloop-science
3
- Version: 0.2.1
3
+ Version: 0.3.0
4
4
  Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
5
5
  Project-URL: Homepage, https://outerloop.science
6
6
  Project-URL: Repository, https://github.com/outerloop-science/outerloop
@@ -0,0 +1,205 @@
1
+ # Accelerators: metering and placement beyond GPUs
2
+
3
+ **Status: proposal, revised after review (2026-10-02); not built.** Outerloop
4
+ meters, caps and places experiment work in GPUs. This note proposes adding
5
+ Cloud TPUs as a second, separately metered device kind, so the same kernel can
6
+ run TPU benchmarks under budgets, operator limits and refusals as strict as
7
+ the GPU ones. Budget caps are safety features: nothing in this proposal may let TPU work
8
+ run unmetered, or let a GPU budget pay for it.
9
+
10
+ ## Why
11
+
12
+ A benchmark that runs on TPUs today would have to declare `gpus: 0`. The
13
+ kernel would then charge no device-hours, apply no `gpu_hours_per_run` budget,
14
+ count nothing against `max_gpus`, and might dispatch its evals as plain CPU
15
+ jobs. The meter is the safety feature, so a new device kind has to be metered,
16
+ placed and capped before any TPU contract is accepted.
17
+
18
+ ## What assumes a GPU today
19
+
20
+ - **Declaration.** `Benchmark.gpus: int (0..8)`; `Measure.gpus`,
21
+ `JobSpec.gpus`; the GPU type lives on the operator's per-target lane
22
+ (`OUTERLOOP_GPU_LANES`, `gpu_lanes.py`). Whether evals dispatch depends on
23
+ `eval_minutes`, not on `gpus`.
24
+ - **Metering.** Launches charge minutes × array × gpus / 60; gates charge the
25
+ main eval (one when a cached baseline applies, else two) and every suite
26
+ sibling pair × walltime × gpus (`syscall.py`). The run meter is
27
+ `stage.gpu_hours_used`; budgets are `gpu_hours_per_run` plus
28
+ `review_topup.gpu_hours`; `max_concurrent_gpus` paces sweeps.
29
+ - **Placement.** `JobSpec.to_argv` emits `--gres=gpu:TYPE:N` or
30
+ `--gpus-per-node=N`, single node; dispatched evals add `--nv` and size CPUs
31
+ and memory per GPU.
32
+ - **Operator limits.** `limits.toml` `max_gpus`, counted from one
33
+ `squeue --me --json` snapshot by parsing `gres/gpu` TRES, attributing jobs to
34
+ this fleet by run id.
35
+ - **Local backend.** `nvidia-smi` detection (or `OUTERLOOP_LOCAL_GPUS`),
36
+ `CUDA_VISIBLE_DEVICES` pinning.
37
+ - **Surfaces.** Briefs, wake lines, refusals, `budget.json`
38
+ (`gpu_hours_remaining`), `outerloop status`, `outerloop limits` and the board
39
+ carry GPU-named fields.
40
+
41
+ ## Proposal
42
+
43
+ The first version is deliberately narrow: **existing GPU keys and behaviour
44
+ stay exactly as they are**, and TPU is added beside them with its own fields.
45
+
46
+ ### Declaration
47
+
48
+ ```yaml
49
+ benchmarks:
50
+ - name: speedrun-tpu
51
+ tpu_chips: 8 # one TPU node per job; mutually exclusive with gpus
52
+ ```
53
+
54
+ `tpu_chips` is a positive integer equal to one supported single-host node
55
+ size (4 or 8 chips in the first version); other values are refused at contract
56
+ load. `tpu_chips` and `gpus` cannot both be set. A suite gate's device demand is
57
+ the union over all its benchmarks, including the one that initiated it. In the
58
+ first version a gate may demand at most one device kind, and the initiating
59
+ benchmark must carry that kind: GPU and TPU never mix in one gate, and a
60
+ CPU-led gate with an accelerator sibling is refused at contract load. That
61
+ kind decides authorization, dispatch and the meter for the whole gate.
62
+
63
+ ### Budgets and meters, per kind
64
+
65
+ - New contract field `budgets.tpu_chip_hours_per_run` (and
66
+ `review_topup.tpu_chip_hours`). A TPU benchmark needs an explicit TPU budget;
67
+ a contract with `tpu_chips` but no TPU budget is refused. GPU budgets never
68
+ authorize TPU spending, and vice versa.
69
+ - New run-stage meter `stage.tpu_chip_hours_used`, retained everywhere
70
+ `gpu_hours_used` is retained today (`STAGE_RETAINED_KEYS` and the stage
71
+ reconstruction paths), and a persisted `stage.meter_kind`. The GPU meter is
72
+ unchanged and stays authoritative for GPU runs, so a rollback never resets GPU
73
+ spending.
74
+ - The arithmetic matches GPUs: chip-hours = minutes × array × chips / 60 for
75
+ launches, evals × walltime × chips for gates.
76
+ - No conversion between kinds and no money in the kernel; operators set
77
+ per-kind caps.
78
+
79
+ ### Placement
80
+
81
+ One TPU node per job in the first version (a single-host slice, e.g. 8 chips):
82
+ the backend requests one whole node from the operator's TPU lane, with no GPU
83
+ GRES and no `--nv`. Multi-host slices are out of scope until the kernel owns
84
+ distributed worker start-up, topology and per-node environment. The TPU lane
85
+ is per target, like GPU lanes, and carries the node's chip count and the
86
+ container image; the kernel refuses a benchmark whose `tpu_chips` does not
87
+ equal the lane's node size.
88
+
89
+ The first backend provisions one Cloud TPU VM per job (Slurm on Google Cloud
90
+ does not support current TPU generations out of the box). It is a bounded
91
+ adapter behind general compute interfaces the kernel gains once: accelerator
92
+ demand, opaque job ids, lifecycle and accounting observations, artifact
93
+ staging, and capability refusals. Cloud API calls, state translation,
94
+ transport and cleanup stay inside the adapter, which owns deterministic
95
+ resource names, reconciliation after an ambiguous create, queue and execution
96
+ deadlines, and deletion confirmed independently of the worker. Author sessions
97
+ and wakes never run on a TPU allocation; the deployment gives them a CPU lane.
98
+
99
+ The TPU worker runs the benchmark in the same no-credential jail as dispatched
100
+ GPU evals: the adapter's cloud credentials, the GitHub App key and any operator
101
+ secret stay in the adapter's process and never reach the TPU VM or the
102
+ container. The VM's own service identity can only read the job's inputs and
103
+ write its outputs in the staging bucket; it cannot create, change or delete
104
+ TPU resources.
105
+
106
+ ### Operator limits
107
+
108
+ `limits.toml` gains `max_tpu_chips` (fleet default and per target). Admission
109
+ checks every kind a request demands; a ceiling on one kind is not skipped
110
+ because another kind has none. The kernel persists what it submitted for each
111
+ TPU job (allocation id, chips, array concurrency) and counts running and
112
+ pending allocations from that record, reconciled with the backend. A fleet
113
+ job whose TPU allocation cannot be bounded from either source blocks further
114
+ TPU admission until it is resolved; it is never counted as zero or as a
115
+ guess. Admission is serialized: a request takes the fleet's admission lock,
116
+ counts persisted allocations plus outstanding reservations, writes its own
117
+ reservation (allocation id, chips) before asking the backend to create
118
+ anything, and releases the lock; the reservation turns into an allocation or
119
+ is removed on refusal or failure. Two concurrent requests therefore cannot
120
+ both fit into the same headroom. A finite ceiling with no usable snapshot fails closed, as today. A
121
+ capacity refusal parks the run (#453). The limits bound this fleet's own jobs
122
+ (attributed by run id); they are not a cloud-project-wide cap, which the
123
+ deployment enforces separately.
124
+
125
+ ### Refusal until complete
126
+
127
+ TPU support ships behind one operator setting, `OUTERLOOP_TPU=1` in the
128
+ operator settings file, read by the tick like the other settings. It defaults
129
+ to off, and the kernel refuses to enable it unless a TPU lane and a finite
130
+ `max_tpu_chips` are configured. Until declaration, budgets, meters, placement,
131
+ limits, caches and surfaces are all in place, the setting is not offered. Until then a
132
+ contract with `tpu_chips` is refused with a clear message, never run as CPU.
133
+ The local backend refuses TPU benchmarks explicitly.
134
+
135
+ ### Caches
136
+
137
+ Both measurement caches (the baseline cache and the dispatched-eval
138
+ determinant) include the device kind, count, resolved image and runtime
139
+ identity. A legacy cache entry never satisfies a TPU measurement.
140
+
141
+ ### Preemption
142
+
143
+ Spot TPU and spot GPU nodes can be reclaimed. The first version submits TPU
144
+ jobs with no automatic requeue: a preempted job fails, the author sees it on
145
+ wake, and its time is charged. Automatic retries come later and only with
146
+ cumulative accounting (every execution's time charged, the retry reserved
147
+ before it starts).
148
+
149
+ ### Surfaces
150
+
151
+ Text names the unit in front of the reader ("GPU-hours", "TPU chip-hours")
152
+ from one helper. `budget.json`, `outerloop status` and the board gain
153
+ `tpu_chip_hours_*` fields beside the GPU ones; durable board rows record the
154
+ meter kind with the value.
155
+
156
+ ## Existing GPU accounting gaps (separate fixes, found in review)
157
+
158
+ - The dispatched-eval cache determinant omits the GPU count, so a result can
159
+ be reused across different GPU counts (`measure.py`).
160
+ - Jobs allocate declared minutes plus ten setup minutes, but only the declared
161
+ minutes are charged (`dispatch.py` `eval_job_spec`); refunds never debit an
162
+ overrun.
163
+ - Launch-hour reconciliation reads one `sacct` record, so a requeued job's
164
+ earlier executions can drop out of accounting.
165
+ - `max_concurrent_gpus` is a per-launch pacing hint (at least one task runs),
166
+ not an aggregate cap; concurrent admissions can overshoot `max_gpus`
167
+ (documented in `operator_limits.submit_batch`).
168
+
169
+ These are fixed for GPUs first (they are the template the TPU meter copies),
170
+ each in its own PR.
171
+
172
+ ## Compatibility
173
+
174
+ New persisted fields only: `tpu_chips` and TPU budget fields in contracts,
175
+ `tpu_chip_hours_used` and `meter_kind` in run stages, the per-job TPU
176
+ allocation and reservation records (allocation id, chips, array concurrency)
177
+ under the run and the fleet state root, `max_tpu_chips` in `limits.toml`,
178
+ device identity in cache keys. GPU records, contracts and limits
179
+ are byte-identical. Rolling back: an older kernel rejects contracts and
180
+ `limits.toml` files carrying the new keys (unknown fields are errors there).
181
+ Before rolling back: end every TPU run, parked ones included (`outerloop end`),
182
+ and confirm no TPU allocation remains; then restore contracts without
183
+ `tpu_chips`, `budgets.tpu_chip_hours_per_run` and `review_topup.tpu_chip_hours`,
184
+ and `limits.toml` without `max_tpu_chips`. Older kernels load ended TPU run
185
+ records (unknown fields are ignored there) and show them without TPU usage.
186
+ GPU runs are unaffected. Release fixtures cover legacy records, an interrupted
187
+ TPU run, and an idempotent rollback.
188
+
189
+ ## Phases
190
+
191
+ 0. **GPU accounting fixes** above (independent PRs).
192
+ 1. **TPU, end to end behind the flag**: declaration, budgets, meters, limits,
193
+ caches, placement on one backend, surfaces, refusals; fixtures for legacy
194
+ records and contracts.
195
+ 2. **First TPU fleet**: one TPU benchmark, one smoke run, then real runs.
196
+ 3. Later, only if needed: multi-host slices, automatic retries with
197
+ cumulative accounting, local TPU.
198
+
199
+ ## Decisions
200
+
201
+ 1. Cross-kind work in one run: no. A benchmark has one kind; suites are
202
+ homogeneous.
203
+ 2. Money in the kernel: no. Per-kind caps; cost reporting stays operational.
204
+ 3. Preemption: fail and report on wake, all time charged; retries later with
205
+ cumulative accounting.
@@ -0,0 +1,95 @@
1
+ # Environment, not workflow
2
+
3
+ Status: proposal, 2026-09-25.
4
+
5
+ ## The rule
6
+
7
+ The kernel adds structure only where it guards a trust boundary. Everything
8
+ else is environment: the agent uses its tools as it sees fit, and research
9
+ habits are taught through the brief and skills, where they can change without
10
+ a kernel release and where each adopter can set their own.
11
+
12
+ The boundaries:
13
+
14
+ - **Credentials.** Sessions hold none. Anything published goes through the
15
+ kernel.
16
+ - **Measurement and credit.** The gate, the panel, the ledger, and what
17
+ reaches the base branch.
18
+ - **Budgets.** GPU-hours, launches and sleeps.
19
+ - **Other agents' state.** Another author's branches and private memory.
20
+
21
+ A rule that guards none of these is a convention, not a mechanism. If a
22
+ convention keeps failing in a way that crosses a boundary, it earns a
23
+ mechanism then, not before.
24
+
25
+ ## What this changes now
26
+
27
+ **1. Sessions may commit on their own line.** The rule "do not commit" guards
28
+ no boundary. The sealed tree is what gets measured, the scope check guards
29
+ what reaches the base branch, and the `.git` tamper guard protects the
30
+ repository. The rule did cause harm: an author could not merge the updated
31
+ base branch into its own work until a special permission was added, and
32
+ merging another author's work would have needed another. Lifting it removes that class of exception.
33
+ Pushing stays with the kernel.
34
+
35
+ **2. Authors may publish branches under their own namespace.** One generic
36
+ capability: stage a push of a named branch under `ideas/<agent-id>/`. The
37
+ kernel checks these things and nothing else:
38
+
39
+ - the name stays inside the author's namespace and is a valid ref name;
40
+ - the author's memory (`AGENT_MEMORY.md`, `agent_memory/`) is left out of
41
+ the published tree, and instruction-bearing files carry the base
42
+ branch's reviewed versions, the same cleanup the kernel applies when it
43
+ checks out a line;
44
+ - the author stays within a bound: at most 20 published branches, and a
45
+ push that adds more than 50 MB of new objects is refused. The byte
46
+ bound applies to every push the kernel makes for an author, line
47
+ snapshots included, which today are unbounded. Both numbers are
48
+ defaults the contract can change.
49
+
50
+ Memory is owned, not secret. Every session already fetches every agent
51
+ line, so an author's memory is readable by its siblings today. Leaving it
52
+ out of the published tree keeps it out of merges and measured trees; it
53
+ does not remove it from history. If memory ever needs to be confidential,
54
+ it has to leave the shared repository, which is a separate design.
55
+
56
+ The author may also reset or delete its own branches. Work another author
57
+ already merged survives in that author's line; only the shared name goes.
58
+
59
+ Publishing costs no budget and earns no credit. A failed publish comes back
60
+ to the author as a message.
61
+
62
+ Branches under another agent's namespace are read-only to everyone else.
63
+ Every session already fetches all branches, so reading and merging them is
64
+ plain git.
65
+
66
+ **3. Ideas are a convention, not a mechanism.** How an author names an idea,
67
+ describes it, marks it parked or abandoned, and finds a sibling's idea is up
68
+ to the author, using git. The brief and a skill suggest a default: the
69
+ branch name says what the idea is, the commit message states the mechanism
70
+ and points at the reports, and a PR that builds on another author's idea
71
+ cites its branch and commit. The board lists the branches under `ideas/`
72
+ with their last commit, and nothing more.
73
+
74
+ ## Two protections a merge needs
75
+
76
+ When a session merges another branch into its line, the kernel keeps the
77
+ receiver's private memory and the reviewed instruction files. It already
78
+ resets instruction files after merging the base branch at run start; keeping
79
+ the receiver's memory through an arbitrary merge is new, since today's rule
80
+ for it covers only resets. The scope check is unchanged: code merged from an
81
+ idea counts as the author's change when it reaches the base branch.
82
+
83
+ ## Deferred
84
+
85
+ Idea status fields, automatic parking, per-author caps, "best number"
86
+ tracking, a separate discovery command, and an operator command for humans.
87
+ Each can be added if the plain version shows a need.
88
+
89
+ ## Decided
90
+
91
+ - Authors may reset or delete their own published branches (owner,
92
+ 2026-09-25).
93
+ - The namespace is `ideas/<agent-id>/`: it names the purpose and matches
94
+ the "one idea per PR" language of the brief and panel. Adopters may use
95
+ it for other purposes; the kernel only enforces ownership.
@@ -94,7 +94,7 @@ park, terminal at the ending. Legs are turns inside it.
94
94
  An ending is a completion with an outcome, not a task state. A negative
95
95
  result is successful work; a rejected PR is a human's decision, not the
96
96
  agent rejecting the task. Only kernel-side failure (stuck, aborted) maps to
97
- failed or canceled. The six endings travel as data on the final message.
97
+ failed or canceled. The endings travel as data on the final message.
98
98
 
99
99
  | Outerloop | A2A | Note |
100
100
  | --- | --- | --- |
@@ -98,7 +98,7 @@ needs inbound access is a human with the runbook.
98
98
 
99
99
  ## Task granularity
100
100
 
101
- A task is **one hypothesis, one PR** — but its evaluation scope follows the
101
+ A task is **one idea, one PR** — but its evaluation scope follows the
102
102
  target's artifact structure, declared in the contract:
103
103
 
104
104
  - *Independent solvers* (the pilot): scope = a single benchmark; one PR moves
@@ -114,8 +114,9 @@ Scope checks compare the full candidate tree with its merge-base against the fet
114
114
  base tip, so changes made only on the base branch do not count as author edits.
115
115
  Missing shared history refuses publication; the separate base-tip ancestry check still applies.
116
116
 
117
- What stays absolute regardless of scope: one hypothesis per PR (attributable
118
- diffs), full-scope reporting, and **cross-target separation** — separate clones,
117
+ What stays absolute regardless of scope: one idea per PR (an idea may bring
118
+ the few changes it needs, and the report attributes each one), full-scope
119
+ reporting, and **cross-target separation** — separate clones,
119
120
  branches, budgets, and report streams per target (`runs/<target>/`,
120
121
  `lessons/<target>.md`), never cross-target code reuse. The only global artifact
121
122
  is this machinery; the only cross-project channel is process lessons in the