outerloop-science 0.1.1__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (200) hide show
  1. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/CHANGELOG.md +95 -0
  2. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/CITATION.cff +2 -2
  3. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/PKG-INFO +1 -1
  4. outerloop_science-0.2.0/docs/design/agent-protocols.md +401 -0
  5. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/agent-substrate.md +4 -4
  6. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/architecture.md +14 -9
  7. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/base-reintegration.md +7 -7
  8. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/consolidation.md +1 -1
  9. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/dispatcher.md +5 -5
  10. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/headline.md +3 -5
  11. outerloop_science-0.2.0/docs/design/lifecycle.md +359 -0
  12. outerloop_science-0.2.0/docs/design/multi-cluster.md +464 -0
  13. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/orchestrator-verify.md +13 -79
  14. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/research-lines.md +6 -1
  15. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/research-loop-buildout.md +23 -1
  16. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/research-loop.md +4 -3
  17. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/reviewer-infra.md +4 -2
  18. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/roles.md +9 -8
  19. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/install.md +185 -19
  20. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/roadmap.md +3 -1
  21. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/validation/author-syscalls.md +1 -1
  22. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/pyproject.toml +5 -0
  23. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/README.md +2 -2
  24. outerloop_science-0.2.0/scripts/install_claude.sh +82 -0
  25. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/tick_chain.sbatch +5 -12
  26. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/tick_deploy.sh +30 -11
  27. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/tick_resident.sh +3 -1
  28. outerloop_science-0.2.0/src/outerloop/__init__.py +3 -0
  29. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/appauth.py +4 -2
  30. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/appmanifest.py +7 -0
  31. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/attempt.py +2187 -931
  32. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/brief.py +58 -106
  33. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/cli.py +205 -53
  34. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/climbboard.py +81 -38
  35. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/compute.py +414 -39
  36. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/contract.py +38 -25
  37. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/contract_cli.py +1 -7
  38. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/disk.py +1 -1
  39. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/dispatch.py +15 -2
  40. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/github.py +97 -47
  41. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/harness.py +55 -11
  42. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/housekeeping.py +2 -3
  43. outerloop_science-0.2.0/src/outerloop/hypothesis.py +45 -0
  44. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/image.py +9 -9
  45. outerloop_science-0.2.0/src/outerloop/inbox.py +752 -0
  46. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/init.py +186 -58
  47. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/intake.py +43 -16
  48. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/measure.py +51 -9
  49. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/orchestrator.py +313 -148
  50. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/panel.py +4 -20
  51. outerloop_science-0.2.0/src/outerloop/paths.py +31 -0
  52. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/progress.py +1 -1
  53. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/role_runner.py +2 -1
  54. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/roles.py +0 -38
  55. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/rolespec.py +1 -3
  56. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/runstate.py +256 -46
  57. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/steward.py +58 -27
  58. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/syscall.py +140 -181
  59. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/syscall_cli.py +204 -51
  60. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/tick.py +455 -485
  61. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/verifier.py +2 -2
  62. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/watcher.py +1 -1
  63. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/conftest.py +15 -0
  64. outerloop_science-0.2.0/tests/test_app_permissions.py +347 -0
  65. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_appauth.py +27 -0
  66. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_appmanifest.py +3 -0
  67. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_attempt.py +1278 -234
  68. outerloop_science-0.2.0/tests/test_attempt_review.py +1424 -0
  69. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_bot_aliases.py +1 -1
  70. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_bot_login.py +2 -2
  71. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_brief.py +47 -27
  72. outerloop_science-0.2.0/tests/test_channel_dir.py +32 -0
  73. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_climbboard.py +234 -22
  74. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_compute.py +9 -1
  75. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_contract.py +31 -3
  76. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_contract_cli.py +2 -2
  77. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_contract_names.py +15 -26
  78. outerloop_science-0.2.0/tests/test_default_claude_model.py +95 -0
  79. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_dispatch.py +44 -0
  80. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_github.py +134 -76
  81. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_hardening.py +2 -2
  82. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_harness.py +98 -18
  83. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_hermes_harness.py +2 -1
  84. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_housekeeping.py +3 -3
  85. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_image.py +24 -4
  86. outerloop_science-0.2.0/tests/test_inbox.py +901 -0
  87. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_init.py +100 -19
  88. outerloop_science-0.2.0/tests/test_install_harness.py +218 -0
  89. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_intake.py +30 -0
  90. outerloop_science-0.2.0/tests/test_lifecycle_cli.py +23 -0
  91. outerloop_science-0.2.0/tests/test_local_compute.py +1060 -0
  92. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_measure.py +100 -0
  93. outerloop_science-0.2.0/tests/test_messages.py +711 -0
  94. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_orchestrator.py +199 -99
  95. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_panel.py +16 -5
  96. outerloop_science-0.2.0/tests/test_paths.py +23 -0
  97. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_role_runner.py +0 -11
  98. outerloop_science-0.2.0/tests/test_runstate.py +485 -0
  99. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_start.py +299 -66
  100. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_steward.py +21 -5
  101. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_syscall.py +108 -132
  102. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_syscall_cli.py +147 -6
  103. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_tick.py +1147 -1304
  104. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_tick_resident.py +24 -9
  105. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_version.py +5 -0
  106. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_watcher.py +1 -3
  107. outerloop_science-0.1.1/src/outerloop/__init__.py +0 -18
  108. outerloop_science-0.1.1/src/outerloop/followup.py +0 -2172
  109. outerloop_science-0.1.1/src/outerloop/paths.py +0 -40
  110. outerloop_science-0.1.1/tests/test_channel_dir.py +0 -35
  111. outerloop_science-0.1.1/tests/test_env_bridge.py +0 -86
  112. outerloop_science-0.1.1/tests/test_followup.py +0 -3128
  113. outerloop_science-0.1.1/tests/test_local_compute.py +0 -351
  114. outerloop_science-0.1.1/tests/test_paths.py +0 -29
  115. outerloop_science-0.1.1/tests/test_runstate.py +0 -270
  116. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/.gitignore +0 -0
  117. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/.pre-commit-config.yaml +0 -0
  118. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/.python-version +0 -0
  119. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/CLAUDE.md +0 -0
  120. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/CONTRIBUTING.md +0 -0
  121. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/LICENSE +0 -0
  122. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/NOTICE +0 -0
  123. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/README.md +0 -0
  124. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/RELEASING.md +0 -0
  125. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/SECURITY.md +0 -0
  126. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/containers/README.md +0 -0
  127. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/containers/agent-py312.def +0 -0
  128. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/assets/icon-dark.svg +0 -0
  129. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/assets/icon-light.svg +0 -0
  130. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/assets/icon.svg +0 -0
  131. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/community.md +0 -0
  132. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/compute.md +0 -0
  133. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/contract.md +0 -0
  134. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/eval-cache.md +0 -0
  135. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/external.md +0 -0
  136. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/github-app-auth.md +0 -0
  137. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/judge-placement.md +0 -0
  138. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/meta.md +0 -0
  139. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/onboarding.md +0 -0
  140. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/public-surface.md +0 -0
  141. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/resident-tick.md +0 -0
  142. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/review-placement.md +0 -0
  143. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/role-cli.md +0 -0
  144. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/scaling.md +0 -0
  145. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/design/session-watcher.md +0 -0
  146. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/docs/reviewer.md +0 -0
  147. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/examples/review.yml +0 -0
  148. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/install_codex.sh +0 -0
  149. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/install_hermes.sh +0 -0
  150. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/requeue_moved_successors.sh +0 -0
  151. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/setup_branch_protection.sh +0 -0
  152. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/scripts/sweep_git_locks.sh +0 -0
  153. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/__main__.py +0 -0
  154. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/evalcache.py +0 -0
  155. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/launchlog.py +0 -0
  156. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/limits.py +0 -0
  157. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/maintain.py +0 -0
  158. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/maintain_agent_cli.py +0 -0
  159. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/maintain_post_cli.py +0 -0
  160. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/markers.py +0 -0
  161. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/posting.py +0 -0
  162. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/py.typed +0 -0
  163. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/review.py +0 -0
  164. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/review_agent.py +0 -0
  165. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/review_agent_cli.py +0 -0
  166. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/review_post_cli.py +0 -0
  167. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/review_summarize_cli.py +0 -0
  168. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/style.py +0 -0
  169. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/verify_agent.py +0 -0
  170. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/verify_agent_cli.py +0 -0
  171. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/src/outerloop/verify_post_cli.py +0 -0
  172. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/fakes.py +0 -0
  173. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/helpers.py +0 -0
  174. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_codex_harness.py +0 -0
  175. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_disk.py +0 -0
  176. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_evalcache.py +0 -0
  177. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_import.py +0 -0
  178. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_launchlog.py +0 -0
  179. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_limits.py +0 -0
  180. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_maintain.py +0 -0
  181. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_markers.py +0 -0
  182. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_measure_and_decide.py +0 -0
  183. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_packaging.py +0 -0
  184. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_posting.py +0 -0
  185. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_progress.py +0 -0
  186. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_requeue_moved_successors.py +0 -0
  187. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review.py +0 -0
  188. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review_agent.py +0 -0
  189. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review_agent_cli.py +0 -0
  190. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review_hardening.py +0 -0
  191. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review_policy.py +0 -0
  192. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_review_summarize.py +0 -0
  193. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_rolespec.py +0 -0
  194. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_sweep_git_locks.py +0 -0
  195. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_tick_chain_successors.py +0 -0
  196. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_tiers.py +0 -0
  197. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_verifier.py +0 -0
  198. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_verify_agent.py +0 -0
  199. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/tests/test_verify_agent_cli.py +0 -0
  200. {outerloop_science-0.1.1 → outerloop_science-0.2.0}/uv.lock +0 -0
@@ -6,6 +6,101 @@ Versions follow [SemVer](https://semver.org).
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.2.0] - 2026-09-18
10
+
11
+ **Upgrade note.** After upgrading run `outerloop permissions --open`: the App needs `checks: read` and `actions: read` for check results to reach authors as messages. This release redesigns the run lifecycle. It adds three run states, one inbox, `end`, review top-ups, and wakes that run by default. It also removes the old `AUTORESEARCH_*` names.
12
+
13
+ Authors now use `message` for public posts, reminders to self and messages to
14
+ live agents on the same target. Sibling messages keep a sent copy, and inbox
15
+ headers name both parties with local message numbers. `--reply-to` links a
16
+ response; `message --show` reads its chain. The kernel bounds delivery and
17
+ reports refused messages. The old reply and note verbs are removed.
18
+
19
+ Messages carry a global id, a context, a recipient and an optional reply reference;
20
+ old inbox files read as before; the wake's headers name the sender.
21
+
22
+ Local mode now shares workstation GPUs across jobs first come first served and runs
23
+ launch arrays in parallel. The board shows running GPU jobs. `OUTERLOOP_LOCAL_GPUS`
24
+ overrides GPU detection; `0` disables allocation. Submissions still wait for all
25
+ tasks to finish.
26
+
27
+ ### Added
28
+
29
+ - `OUTERLOOP_CLAUDE_MODEL` configures the shared default for all Claude roles, including deployments whose Vertex project has access to a different model. `OUTERLOOP_AUTHOR_MODEL` still overrides the author default. Vertex auxiliary fast calls default to the session model, avoiding dependencies on models the project has not enabled; `OUTERLOOP_VERTEX_SMALL_MODEL` overrides that default.
30
+
31
+ - `outerloop init` installs a missing author CLI and records its path; `--no-install-harness` opts out. Claude has a pinned, SHA256-verified installer.
32
+
33
+ - `OUTERLOOP_TICK_HOST=login` (or `--tick-host login`) runs a niced foreground tick loop against Slurm, with one tick lease per state root.
34
+ - `OUTERLOOP_QOS` sets the QOS for every submitted Slurm job.
35
+ - `OUTERLOOP_APPTAINER_BIN` selects the container executable on compute nodes.
36
+ - `<root>/HOLD_LAUNCHES` holds fresh kernel launches while existing runs, wake delivery, and the tick chain continue; remove the file to resume.
37
+
38
+ ### Removed
39
+
40
+ - The pre-rename names are gone. The `AUTORESEARCH_*` environment names are no longer read by the package, the `.env` reader, the chain or the deploy step. `~/.config/autoresearch/`, `.autoresearch.yaml`, the `.autoresearch` channel directory, the `~/.autoresearch` local root default and `~/autoresearch-images` are no longer looked for. The `autoresearch-resident` and `autoresearch-tick` job names are no longer recognized. The `harness_key` file, the `*_HARNESS_KEY_FILE` setting and the `climb_job_id` record field are gone. Before upgrading, rename the `.env` keys, the config directory and the image directory. Cancel a resident or chain still queued under an old job name (`scancel --name autoresearch-resident`, or `scancel --name autoresearch-tick`) before running `outerloop start`: the guard against two loops on one root knows only the current name. Old `autoresearch:` comment markers are still recognized.
41
+
42
+ ### Fixed
43
+
44
+ - The measurer waits up to 90 seconds for the files of a finished evaluation job. A filesystem delay no longer ends a measured improvement as a negative result.
45
+ - `outerloop start` refuses to run when the configured author's CLI is missing: it checks the recorded path, else PATH, else `~/.local/bin`, and names the install command and `outerloop init --force`.
46
+
47
+ - Ending line snapshots push once, with the kernel's GitHub auth. Without auth nothing is pushed and the log says so. A refused push is retried only when the remote line moved.
48
+ - PRs under `merge: auto` say why self-merge is waiting, and the publish records and logs the bless decision.
49
+
50
+ - A timed-out local job could leave its GPUs reserved until the next submit; they are now released once its process group is gone.
51
+
52
+ - The tick log says why a blessed PR is not merged yet (a message waits for the author, the run sleeps on jobs, the base moved, and so on), one line per sweep.
53
+ - Runs retain their hypothesis through review. The status strip and sibling view carry it with the PR link, and the brief asks authors to check the view, refreshed at every wake, before choosing a direction and before submitting.
54
+
55
+ - The board's squeue card scrolls sideways inside its box on a narrow screen instead of overflowing the page; the job name column is narrowed first.
56
+ - Board rows now keep each run's whole hypothesis paragraph; the site's ledger showed sentences cut at 160 characters. Text over 1,000 characters ends at a word with an ellipsis. The next publish repairs rows cut by the earlier limit.
57
+
58
+ - `outerloop permissions` checks App access and opens the next page to edit or accept missing permissions with `--open`; upgrades, starts and sweep warnings provide guidance for existing installations.
59
+
60
+ - In auto mode the sweep merges only a clean PR at the head the kernel measured and approved. Publish never arms GitHub auto-merge; the sweep withdraws old arms and reports head moves to the author.
61
+
62
+ - A reply returning after PR merge or close preserves the run’s ending. GitHub outages no longer skip job and deadline handling, failed reply posts suppress duplicate final text, and terminal snapshot cleanup survives notebook failures. Inbox delivery retries refused messages and refuses symlinked destinations.
63
+
64
+ - A review panel skipped at preflight or in the follow-up now sends the author a kernel note and records the reason in the submit’s PR addendum. These publishes never arm auto-merge.
65
+
66
+ - A submitted review measurement could go unreported when its push failed. The number is now posted once, as soon as it is known. Review submits push a bot commit with the report’s first line and measured number as its message.
67
+ - On local compute an author-sleep wake could lose an improvement: the gate answers inline there, and the wake only knew how to end a run, so the result became an aborted ending with no PR (cluster0, three runs). Fresh and resumed climbs now share one terminal: the report, the line notebook, a PR or an ending record, and the issue note. A wake that opened its PR but died before recording it reconciles to that PR instead of opening a second, and a park's snapshot is released only once the run has left waiting.
68
+ - A research line lost its snapshot when the author's session reset its branch to main: the seal parented on main, the push was refused, and the run left no notebook entry (#368). The kernel now records the line head itself and the line only moves forward from that record. Memory files a reset dropped from the tree come back at the seal; files the session deleted on the line stay deleted.
69
+ - Intake dropped an issue from a private org member without a word: the App's token sees such an author as CONTRIBUTOR, and only OWNER/MEMBER/COLLABORATOR qualified. Every skipped issue is now logged with its reason, a maintainer's `outerloop:task` label vouches for an issue regardless of association, and new Apps request members read so private members read as MEMBER.
70
+
71
+ ### Changed
72
+
73
+ - Completed CI checks reach the author as messages with bounded log tails. Failed checks wake idle runs; successful, neutral and skipped checks wait as context.
74
+
75
+ - Messages carry an origin and a qualified repository thread. Staged replies keep their destination through retries and later PR changes.
76
+
77
+ - Memory guidance now covers every session boundary and asks authors to record durable findings as they learn them. Sleep, submit and end help say the session ends there.
78
+
79
+ - The board's live strip says what a parked run waits on beside its state: jobs, gate, review or wake. Read off the record; the three states are unchanged.
80
+ - Runs have three states: running, parked and ended; the board and the logs show those names, and an open PR is a link on the run. The sweep delivers comments, base moves and job results through one wake path; the separate follow-up job is gone. Old state names are mapped on read; inbox cursors migrate once under the run lease, and every tick logs the number of legacy follow-up records until it is zero. An older kernel cannot read the new state names; that is the only incompatible field, and fleets are updated by commit. The contract's `followup_job_minutes` now sizes the wake job of a run with an open PR.
81
+
82
+
83
+ - Authors can stage `end [--report <file>]` and end their turn: without a PR it ends as a negative result with the last failed verdict's note, or “ended without a submit”; with an open PR it posts the supplied report and parks until a message arrives. End costs nothing and cannot accompany launch, sleep or submit.
84
+ - Opening a run's first PR grants a review top-up once: 2 launches, 4 sleeps and 0.5 GPU-hours by default, configurable through `budgets.review_topup`. The increased ceilings and prior spend appear in the tool, brief and every wake.
85
+
86
+ - Submit works in review and fast-forwards the PR after a confirmed auto-merge disarm; the measured number is posted first, even if publication is refused. Each review leg measures against its freshly fetched base. Verdicts, findings and publish refusals reach the author as messages. A reply staged during a review leg suppresses its final-text comment. A failed submitted park ends as negative-result only when no author session can resume to receive its verdict; its report and notebook are saved and its issue claim released. Sessions without the tool (no launcher or no resume support) are still measured at finish and may end on the verdict by design. Submit needs no prior launch or report. A session offered submit that stops without it ends unmeasured, or returns to review if it has a PR. Legacy follow-up re-measures retire on their next wake and release their snapshots.
87
+
88
+ - Authors can post replies through `message` and launch experiments or sleep while a PR is in review, using the run’s remaining budget. Comments and base moves received while parked reach the author at its next wake. Replies are kept in an outbox for retries, review launches check committed edits, and closing a run holds its wake lease. Review edits are measured and published only on submit, using the same remaining GPU budget as other author work.
89
+
90
+ - Wake messages now use one inbox and one renderer. Launch results, gate verdicts, panel findings, review comments and base moves reach the session in the same fenced format, from files kept beside the run's record. Advisory panel findings now reach the author alongside blocking ones. The sibling view is refreshed at every wake of a parked author session. The kernel's wake text states facts; the research advice it used to carry is gone.
91
+
92
+ - Dispatched wakes are on by default. The old on-switch (`OUTERLOOP_DISPATCH_WAKE=1` or a `DISPATCH_WAKE` sentinel) is gone; the operator turns wakes off with `OUTERLOOP_DISPATCH_WAKE=0` or a `<root>/DISARM_WAKE` sentinel, and a dry sweep says so in the log. An unarmed loop stranded every parked run silently, which local compute, able to park since 0.1.2, hit at once.
93
+ - The brief's launch section says that a launch runs a sealed snapshot of the working tree and is scope-checked like the final tree, so experiment scripts must live under the contract's allowed paths.
94
+
95
+ - Whether jobs need a lane is now the compute backend's word (`has_lanes`): the measurer, the launcher, and the tick's GPU preflight ask it instead of testing for local mode. The dispatch settings always exist (a backend always does), so local compute arms the author's `launch`/`sleep` syscalls for benchmarks with `depth_k`, and a local wake no longer needs a container image. A `--image` path that is not a file is refused at the command line instead of silently degrading the run to inline evaluation.
96
+ - Author sessions start with no GPU visible on any backend (`CUDA_VISIBLE_DEVICES` is empty), so experiments that need one go through `launch`, where they are recorded in the ledger and metered against the GPU budget. This is the bare session's default; a contained session has no GPU device at all, which is the enforcement.
97
+
98
+ ## [0.1.2] - 2026-09-11
99
+
100
+ ### Fixed
101
+
102
+ - A GPU benchmark on local compute aborted at its first gate measure with "no GPU lane is configured". The measure placement now follows the launch placement's rule: local compute has no lanes, the job runs on the machine's own GPUs.
103
+
9
104
  ## [0.1.1] - 2026-09-11
10
105
 
11
106
  ### Fixed
@@ -13,5 +13,5 @@ authors:
13
13
  repository-code: "https://github.com/outerloop-science/outerloop"
14
14
  url: "https://outerloop.science"
15
15
  license: Apache-2.0
16
- version: 0.1.1
17
- date-released: "2026-09-11"
16
+ version: 0.2.0
17
+ date-released: "2026-09-18"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: outerloop-science
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
5
5
  Project-URL: Homepage, https://outerloop.science
6
6
  Project-URL: Repository, https://github.com/outerloop-science/outerloop
@@ -0,0 +1,401 @@
1
+ # Agent protocols: what A2A and MCP would give Outerloop
2
+
3
+ Status: design note, September 2026, reviewed by a second model. A question
4
+ from the owner: the agent-to-agent protocols have settled since we designed
5
+ the lifecycle. Which of them fit, what would adoption cost, and what would it
6
+ buy?
7
+
8
+ The short answer. A2A's data model (a task with a state machine, messages
9
+ made of parts, "input required", artifacts) resembles what the kernel and an
10
+ author already do, closely enough to sketch a mapping but not closely enough
11
+ to call our design a profile of it. Its transport (JSON-RPC over HTTP) does
12
+ not fit Slurm, where nothing serves HTTP and messages are files. The
13
+ position this note proposes: the durable per-run inbox stays the one message
14
+ store; an A2A adapter that feeds it is a candidate for the first
15
+ service-shaped agent (one that runs as a service, with an endpoint of its
16
+ own, rather than as a batch job the kernel starts), and it earns its place
17
+ by that integration, not before.
18
+ MCP is settled for the agent-to-tool layer; where it fits here depends on a
19
+ retriever decision two existing notes disagree on. Nothing needs building
20
+ today.
21
+
22
+ ## The landscape, as of this month
23
+
24
+ **A2A (Agent2Agent).** Started by Google in April 2025, hosted by the Linux
25
+ Foundation since June 2025, version 1.0 in March 2026 and 1.0.1 in May 2026.
26
+ Over 150 organizations back it, and it ships in Azure AI Foundry, Copilot
27
+ Studio, Amazon Bedrock AgentCore and Google Cloud. An agent publishes an
28
+ Agent Card (identity, skills, capabilities, security schemes, endpoint). A
29
+ client sends a Message; the agent may answer with a Message directly or with
30
+ a Task that moves through submitted, working, input-required, auth-required,
31
+ completed, failed, canceled and rejected. Messages carry Parts (text, file,
32
+ structured data) and Tasks carry Artifacts. Long-running work is polled or,
33
+ optionally, pushed to a webhook or streamed; a task that needs something goes
34
+ to input-required and the client sends the next message with the same task
35
+ id. Three bindings: JSON-RPC 2.0, gRPC and HTTP+JSON. Extensions, declared
36
+ by URI in the Agent Card, may add data, methods and transitions, but not new
37
+ task-state values. IBM's ACP joined A2A in August 2025.
38
+
39
+ **MCP (Model Context Protocol).** The agent-to-tool layer, spoken by every
40
+ harness we run (Claude Code, Codex, hermes; the last two also act as MCP
41
+ servers). The 2026-07-28 revision made the core stateless (no session
42
+ handshake; requests carry their capabilities), added multi-round-trip
43
+ requests so a server can ask the client for input without a persistent
44
+ connection, and moved long-running work into a Tasks extension (a durable
45
+ task id that survives client restarts, polling, input-required, cancel).
46
+ Servers expose tools, resources and prompts; the client owns consent.
47
+
48
+ **The rest.** AGNTCY (Cisco, under the Linux Foundation) covers agent
49
+ discovery, messaging and observability, with its Open Agent Schema Framework
50
+ describing agents. Agent Network Protocol and the on-chain identity efforts
51
+ target decentralized markets. A June 2026 analysis of the governance these
52
+ protocols can express finds voting and dissent preservation absent from all
53
+ of them and deliberation, escalation and audit only partial; the authors
54
+ argue that governance is "a missing architectural layer above current
55
+ interoperability standards." That is consistent with our own line, which
56
+ is stronger: protocols carry messages, never authority.
57
+
58
+ ## Outerloop's seams
59
+
60
+ A protocol only matters at a boundary between two parties that could be
61
+ built by different people. Outerloop has five.
62
+
63
+ 1. **Kernel to author session.** The kernel starts a harness session with a
64
+ brief, the session works, stages requests through the syscall tool
65
+ (launch, sleep, submit, message, end), and ends its turn. The kernel runs the
66
+ rigid steps (launch jobs, measure, publish) and wakes the session with the
67
+ results as messages in its inbox. On Slurm the session is a batch job with
68
+ no inbound network; the channel is files in the workspace and under the
69
+ run directory.
70
+ 2. **Kernel to judges.** The panel lenses and the verifier are sessions too,
71
+ run by `run_role`, recording findings and a verdict through the same tool.
72
+ Their output is evidence for the author and the kernel, never acceptance
73
+ of the research.
74
+ 3. **Kernel to humans.** GitHub: PRs, comments, reviews, checks. The inbox
75
+ already carries these as messages, and the outbox posts replies.
76
+ 4. **Kernel to compute.** Jobs on Slurm or the local pool. Not an agent
77
+ boundary; no protocol applies.
78
+ 5. **Kernel to other agents.** A planner assigning directions, an adopter's
79
+ own agent as an author, a steward or maintainer, sub-agent teams inside a
80
+ run. Today this seam does not exist; the lifecycle doc lists it under
81
+ "Later".
82
+
83
+ ## A tentative mapping onto A2A
84
+
85
+ Written down so the next design item can check itself against it, not as a
86
+ compatibility promise. Two choices had to be made that the resemblance does
87
+ not make for us: what a task is, and what an ending is.
88
+
89
+ A task spans a run, not a leg. An A2A task in input-required stays the same
90
+ task after the client answers, while our leg ends at every sleep. So the
91
+ whole run is one task: working while a session runs, input-required at every
92
+ park, terminal at the ending. Legs are turns inside it.
93
+
94
+ An ending is a completion with an outcome, not a task state. A negative
95
+ result is successful work; a rejected PR is a human's decision, not the
96
+ agent rejecting the task. Only kernel-side failure (stuck, aborted) maps to
97
+ failed or canceled. The six endings travel as data on the final message.
98
+
99
+ | Outerloop | A2A | Note |
100
+ | --- | --- | --- |
101
+ | a run | one task, in one context | legs are turns within it |
102
+ | `running` | working | |
103
+ | `parked` on jobs, on a submit, or on the next human message | input-required | the kernel is the client: its next message is the answer (launch results, the verdicts, a human's comment); the reason travels as data; a checkpoint timeout needs no answer and is a kernel-side wake |
104
+ | `ended` | completed (negative result, merged, rejected, budget exhausted) or failed/canceled (stuck, aborted) | the ending is metadata, not a state |
105
+ | inbox message kinds | structured-data parts of a client message | kinds are data inside a part, not new part types |
106
+ | `launch`, `submit`, `message` | what the agent asks for in input-required | see the caveat below |
107
+ | the report, the PR, the line snapshot | artifacts | |
108
+ | budgets and the meter | kernel-owned, delivered as data | A2A has no budgets |
109
+ | the gate, launching, publishing | the client's own work | never delegated |
110
+
111
+ The caveat. Reading the author's verbs as "the agent asks the client for
112
+ work" is a legitimate use of input-required, but A2A defines no semantics
113
+ for a launch, a measurement or a relayed reply. Those are an application
114
+ contract on top of the protocol. So an arbitrary A2A agent could not become
115
+ an author by pointing the kernel at its card; it would have to speak our
116
+ contract inside A2A's envelope. What A2A gives is the envelope and the
117
+ lifecycle vocabulary, which is worth something, and no more than that.
118
+
119
+ Two things do not map and should not. Authority: A2A carries no notion of
120
+ who may merge, spend, or measure; those stay kernel-side. Trust: an agent's
121
+ messages are data. The inbox renderer fences every body; any adapter would
122
+ feed the same renderer.
123
+
124
+ ## Transport: one inbox, adapters in front of it
125
+
126
+ `lifecycle.md` already says the inbox is a durable store with put, list, get
127
+ and conditional put, that delivery happens only at a wake, and that a cloud
128
+ backend changes the store's implementation and nothing else. HTTP JSON-RPC
129
+ is not a store; it is an execution interface. So the honest shape is:
130
+
131
+ - **The inbox stays the one store**, on every backend. It owns replay,
132
+ ordering, first-key-wins deduplication, cancellation and the wake policy.
133
+ - **An A2A adapter feeds it.** For an agent that lives as a service, the
134
+ kernel acts as the A2A client: send with immediate return, poll the task,
135
+ and append what comes back as inbox messages; answer the agent's
136
+ input-required by running the kernel's own steps and sending the next
137
+ message. Streaming and webhooks are optional in A2A and unnecessary for
138
+ wake-based delivery. Inbound deduplication comes from the inbox's key
139
+ rule. Outbound sends need their own care: a crash between a send and the
140
+ record of its task id would resend and start duplicate remote work, so the
141
+ adapter writes a send record (our message id, reused as A2A's message id)
142
+ before it sends. A2A leaves deduplication by message id optional, so on a
143
+ retry the adapter first lists the context's tasks and adopts the one that
144
+ carries that message id; it resends only when none does.
145
+ - **Slurm and local compute keep the file channel** with no adapter at all.
146
+ Nothing serves HTTP there and nothing needs to.
147
+
148
+ ## MCP: a contradiction to resolve first
149
+
150
+ The syscall tool could be presented as an MCP server. `role-cli.md` rejected
151
+ that: the CLI-over-Bash surface keeps Claude Code, Codex and hermes
152
+ identical, runs inside the jail with no server process, and leaves the
153
+ kernel-side validator as the only trust boundary. All three harnesses now
154
+ take MCP servers by config, including headless runs, so the old asymmetry is
155
+ gone, but nothing has shown a benefit worth a server per session and a
156
+ config per backend. The verbs stay a CLI.
157
+
158
+ Where MCP fits is contested by our own notes. `agent-substrate.md` reserves
159
+ a harness-provided MCP surface for the retriever and PR-context read;
160
+ `role-cli.md` plans `retrieve` as another CLI verb and says no MCP server.
161
+ One of them has to yield. The choice is not about protocols: it is whether
162
+ read tools that the kernel owns should be one more verb on the file channel
163
+ or a discoverable server. Neither is built. MCP's Tasks extension would not
164
+ change the lifecycle either way; the kernel already owns waits, parks and
165
+ wakes, so a second durable-task mechanism inside a tool call has no job here.
166
+
167
+ ## If Outerloop decentralizes: kernel to kernel
168
+
169
+ The owner's forward question: suppose every lab runs its own kernel and
170
+ kernels talk to each other. That is the case where a protocol stops being
171
+ speculative, because a kernel is what A2A was designed around: an always-on
172
+ service with an identity, taking tasks and returning artifacts. Sessions on
173
+ Slurm stay files; kernels on the network do not.
174
+
175
+ What kernels would say to each other, and what already carries it:
176
+
177
+ - **Shared research state on one target.** Attempts, hypotheses, results
178
+ and reports for a benchmark several labs work on. This is already
179
+ decentralized and already has a protocol: git. The research-log branch is
180
+ the ledger, PRs are the human-facing channel, and every kernel reads and
181
+ writes the same repository. What is missing for several kernels on one
182
+ repository is coordination, not transport: who publishes the board, and
183
+ per-kernel namespaces in the ledger so two kernels never fight over one
184
+ file.
185
+ - **Delegated work.** A run on kernel A needs an experiment or a measurement
186
+ that kernel B's compute can run. This is a task: a sealed tree, a command,
187
+ a walltime, and back come numbers, logs and an artifact. It is the compute
188
+ seam (`submit`, `status`) crossing a trust boundary, and an A2A task fits
189
+ it directly: kernel A is the client, kernel B the agent; input-required
190
+ covers "send me the data shards"; the result is an artifact.
191
+ - **Independent verification.** Kernel A asks kernel B to re-measure a claim
192
+ it did not produce. Also a task, with a signed verdict as the artifact.
193
+ A2A 1.0's signed Agent Cards give the identity; the protocol does not make
194
+ the measurement honest. That needs attestation or replay, the crux the
195
+ market note already names, and it stays a kernel-owned rigid step on the
196
+ verifying side.
197
+ - **Discovery.** Which kernels exist, which benchmarks they host, what
198
+ compute they offer. Agent Cards carry the description; a registry or the
199
+ target repository itself carries the list. AGNTCY's schema work is the
200
+ candidate for a registry if one is ever needed.
201
+ - **Authors across kernels.** The sibling view across labs on one target is
202
+ the shared ledger again, read through git, not a message between kernels.
203
+
204
+ What a protocol does not solve there, and what would need its own design:
205
+ trust in another kernel's numbers (attestation, replay), credit and budgets
206
+ across kernels (A2A carries none; the Agent Payments Protocol launched
207
+ beside it is the closest thing), and provenance (which kernel measured what,
208
+ signed). Those are the same rigid steps every kernel keeps for itself today,
209
+ extended across a boundary.
210
+
211
+ The consequence for the present design is small and concrete: keep the
212
+ message model A2A-shaped and keep the transport behind the inbox, so that a
213
+ kernel-to-kernel adapter is the same adapter as the service-agent one, with
214
+ each kernel acting as client in one direction and agent in the other. The
215
+ market design (a separate private note) is the first place this would be
216
+ needed; git stays the substrate for everything that is about one repository.
217
+
218
+ ## Decision: an A2A-shaped message model, and coordination inside one kernel
219
+
220
+ The owner's decision (2026-09-14): the message model becomes more A2A-shaped,
221
+ and this is the same effort as multi-agent author coordination inside one
222
+ kernel, which `lifecycle.md` had parked under "Later" and `scaling.md`
223
+ describes as the planner. This section is the design for both, revised
224
+ after a second model's review, which cut it down: stable identities and
225
+ bounded routing first, protocol naming only where it costs nothing, and no
226
+ schema work ahead of an adapter.
227
+
228
+ **Simple messages, not conversations.** There is no conversation object.
229
+ A message is one thing said once; a response is another message that names
230
+ the one it answers. The agent's own session is its memory of what it said
231
+ and heard, resumed at every wake; the inbox is what arrived since. So at a
232
+ wake the agent sees the full inbox view it sees today, everything
233
+ undelivered in arrival order, each item fenced, and a message that answers
234
+ an earlier one says so in one line (`replying to #n`). The wake has no grouping
235
+ or summaries of past exchanges;
236
+ the session already holds those. This is what A2A's context and task ids
237
+ are for as well: correlation, so a reader can tell what a message is about,
238
+ never a structure the reader must reconstruct.
239
+
240
+ **The envelope.** Today a message is (seq, kind, source, origin, thread,
241
+ arrived, key, payload). The changes are small and each has one job:
242
+
243
+ - `message_id`: globally unique, deterministic, namespaced by the inbox
244
+ (`<run id>/<key>`), so a message that crosses an adapter or is quoted by
245
+ another run has one name. Deduplication stays first-key-wins per inbox.
246
+ - `context_id`: what a message is about, with one rule: the run it concerns.
247
+ A planner's own inbox is its search line's context; a message it sends
248
+ into an author's inbox is about that author's run.
249
+ - `from` and `to`: `from` keeps today's category and identity (`source`,
250
+ `origin`) and the kernel sets it, never the sender; `to` is the recipient
251
+ as a kernel identity (a run id, `kernel`, a GitHub thread) and is what
252
+ routing needs. Replies already store their destination when staged; this
253
+ makes every message do so.
254
+ - `in_reply_to`: optional, the `message_id` this one answers. Correlation,
255
+ nothing more.
256
+ - `kind`, `payload`, `thread`, `arrived`, `seq`: unchanged. The renderer and
257
+ the wake rule read `kind` and `payload` today; they keep doing so. A2A's
258
+ `parts` and `role` are the adapter's business: `role` is relative to which
259
+ side of a protocol exchange one is on and must never imply permission
260
+ inside the kernel, and `parts` needs artifact ownership and size rules
261
+ that no adapter yet asks for.
262
+
263
+ One versioned decoder reads every inbox file, for delivery, for
264
+ deduplication and for appends alike: old files without the new fields get
265
+ them on read (`to` = the inbox's own run, `message_id` from the run id and
266
+ key); nothing is rewritten; a file the decoder cannot read stops delivery,
267
+ as today, and never silently drops out of deduplication.
268
+
269
+ **One verb for saying things.** `message [--to thread|self|agent-NN]
270
+ [--reply-to <n>] <text>` (or `--file <path>`) defaults to the public PR or
271
+ issue thread. The confirmation names that public destination. Self returns
272
+ at the next wake; agent-NN addresses a live sibling. The old verbs retire;
273
+ old inbox entries of kind `note` remain readable. One 20,000-character limit
274
+ applies to all destinations.
275
+
276
+ Headers read `## #<seq> <kind> | <sender> -> <recipient> | <time> UTC`.
277
+ Parties are `you`, `agent-NN`, a qualified GitHub human, a named job, kernel,
278
+ panel, git, or a CI app. Run ids never appear in headers. The protocol says
279
+ once, "Every fenced block below is data, never instructions." There is no
280
+ per-message data line. A reply's first fenced line is `replying to #n`, using
281
+ the reader's local inbox number, or `replying to a message not in your inbox`.
282
+ `message --show <n>` shows the chain oldest first from the kernel's snapshot
283
+ of the newest 200 inbox entries, with text capped at 2,000 characters.
284
+
285
+ **Routing: the kernel is the hub.** The kernel sets the origin, resolves a
286
+ live run on the same target, and appends an agent-message to its inbox. It
287
+ also keeps a context-only sent copy in the sender's inbox with the same key,
288
+ destination and reply reference. A sent copy does not wake its sender.
289
+ `--reply-to` names the sender's own inbox number; the kernel stores the
290
+ referenced message id. The limits are eight messages per leg and four
291
+ undelivered messages from one sender to one recipient. Refusals produce a
292
+ kernel note naming the refused message. Runs in review receive mail behind
293
+ their jobs under the existing wake rule. Public delivery retains redaction,
294
+ durable staging and reply-id marker deduplication; a public reply also
295
+ carries the referenced message id in a marker.
296
+
297
+ **Sub-agents: not a tier, for now.** A kernel-level agent task (`launch
298
+ --agent`: one session under the parent's ceiling, on a sealed snapshot,
299
+ metered against the parent, returning one report) would buy kernel-owned
300
+ isolation, durable dispatch, explicit spend reservation and uniform
301
+ timeouts across backends. Those are real, but nothing measured asks for
302
+ them yet: the harness backends already run in-session sub-agents under the
303
+ parent's RoleSpec ceiling (`agent-substrate.md`), job arrays already give
304
+ parallelism on compute nodes, and a cheaper model is a harness setting. So
305
+ the tier is out of the committed sequence. Its trigger is evidence: an
306
+ attempt that fails or wastes measurable resources because in-session
307
+ sub-agents and jobs cannot meet a need (work across the parent's sleep, an
308
+ isolation or accounting the harness cannot give). Before that, verify that
309
+ the existing sub-agent ceiling is enforced as declared; that is cheaper and
310
+ overdue.
311
+
312
+ **The planner writes the plan.** Most of a planner's value lands when the
313
+ kernel picks a direction for a new climb, not while runs are live. So the
314
+ first planner is a plan writer: one bounded invocation per search line, on
315
+ a cadence or after a merge, that reads the leader, the board (which now
316
+ carries every run's hypothesis), recent reports, lessons and the budget,
317
+ and drafts the plan section of the search-line issue. The kernel posts it
318
+ (the planner never touches GitHub) and puts the plan's open directions,
319
+ beside the current ledger activity, into every new climb's brief as
320
+ advisory context under the existing admission and budget rules. No
321
+ persistent planner inbox or session, no routine messages to live runs, no
322
+ second plan store, no dependency on agent tasks or on the envelope change.
323
+ It holds no authority: no starting runs, no spending, no measuring, no
324
+ merging, no approving its own plan; the issue's veto window stays the human
325
+ gate. Evidence: duplicate hypotheses and duplicated experiment spend against
326
+ the weeks before, on the same benchmark. It cannot guarantee non-overlap
327
+ (two authors can pick the same open direction from identical briefs before
328
+ either claim reaches the ledger); if advisory context proves insufficient,
329
+ an atomic claim at admission is the next step, and live redirection through
330
+ `message --to` only after observed collisions justify it.
331
+
332
+ **Sequencing.** Each stage lands as one PR, reviewed and run on one fleet
333
+ from its commit, with its own acceptance evidence; "nothing moved" is not
334
+ evidence.
335
+
336
+ 1. **Envelope and decoder.** `message_id`, `context_id`, `from`/`to`,
337
+ `in_reply_to`, one versioned decoder, every producer setting them, the
338
+ header naming senders by identity. Fleet evidence: old and new inbox
339
+ files delivered identically, a restart mid-wake, a duplicate append, a
340
+ damaged entry, unchanged wakes.
341
+ 2. **One `message` verb and sibling routing.** `message --to
342
+ thread|self|<agent>`, `reply` and `note` retired, the kernel's validation
343
+ and limits, delivery into the recipient's inbox, the board showing a
344
+ message's sender. Fleet evidence: a public reply still lands once and
345
+ redacted; a note still comes back; one author tells a sibling something
346
+ and the sibling reads it at its next wake; a forged identity, a flood, a
347
+ message in review and one to an ended run each handled as specified.
348
+ 3. **The plan-writing planner.** One search line on gpt-speedrun. Fleet
349
+ evidence: fewer duplicate hypotheses and less duplicated spend than the
350
+ unplanned weeks before; the veto window exercised.
351
+
352
+ Sub-agents wait for their trigger. Stage 1 is mechanical; stages 2 and 3
353
+ are the semantic ones and each is read by the owner before it is built.
354
+
355
+ ## Criteria instead of approvals
356
+
357
+ Nothing in code now. The note itself is the deliverable. What would justify
358
+ each later step:
359
+
360
+ - **An A2A adapter** when a service-shaped agent has a concrete owner and a
361
+ first integration: the cloud backend, an adopter who wants to bring an
362
+ agent with an existing A2A endpoint, the planner from `scaling.md`, or
363
+ another kernel (the market). The
364
+ integration defines the application contract inside A2A's envelope; the
365
+ adapter feeds the inbox; the kernel stays a client and needs no Agent Card
366
+ of its own. Pin the A2A version.
367
+ - **Vocabulary and addressing.** Decided above: the envelope takes A2A's
368
+ names, and routing is the kernel's (`from`, `to`), since A2A's ids are
369
+ correlation, not identity.
370
+ - **The retriever surface.** Decide CLI verb or MCP server on its own merits
371
+ before either is built.
372
+
373
+ Not planned: A2A as the transport on Slurm, an agent directory, signed Agent
374
+ Cards for our fleet, a state-machine extension (A2A cannot add states), and
375
+ any protocol feature that would move authority out of the kernel.
376
+
377
+ ## Questions for the owner
378
+
379
+ 1. Does A2A stay an adapter candidate under the criteria above, with the
380
+ first service-shaped integration deciding it, rather than a schema we
381
+ adopt now?
382
+ 2. Which retriever surface wins, the CLI verb of `role-cli.md` or the MCP
383
+ server of `agent-substrate.md`? The loser's sentence should be struck.
384
+ 3. Is there a first integration on the horizon (cloud backend, an adopter's
385
+ agent, the planner) that should set the timeline?
386
+
387
+ ## Sources
388
+
389
+ - A2A specification 1.0: https://a2a-protocol.org/latest/specification/
390
+ - A2A 1.0.1 release: https://github.com/a2aproject/A2A/releases/tag/v1.0.1
391
+ - A2A extensions (no new state values): https://a2a-protocol.org/latest/topics/extensions/
392
+ - A2A, life of a task: https://a2a-protocol.org/latest/topics/life-of-a-task/
393
+ - ACP joins A2A (August 2025): https://lfaidata.foundation/communityblog/2025/08/29/acp-joins-forces-with-a2a-under-the-linux-foundations-lf-ai-data/
394
+ - Linux Foundation, A2A first year (April 2026): https://www.linuxfoundation.org/press/a2a-protocol-surpasses-150-organizations-lands-in-major-cloud-platforms-and-sees-enterprise-production-use-in-first-year
395
+ - MCP specification 2026-07-28 and changelog: https://modelcontextprotocol.io/specification/latest and https://modelcontextprotocol.io/specification/2026-07-28/changelog
396
+ - MCP Tasks extension: https://modelcontextprotocol.io/extensions/tasks/overview
397
+ - AGNTCY: https://agntcy.org/
398
+ - Kang and Diponegoro, Governance Gaps in Agent Interoperability Protocols (June 2026): https://arxiv.org/abs/2606.31498
399
+ - Claude Code MCP configuration: https://code.claude.com/docs/en/mcp
400
+ - Codex MCP configuration: https://developers.openai.com/codex/mcp
401
+ - hermes MCP: https://hermes-agent.nousresearch.com/docs/user-guide/features/mcp
@@ -61,7 +61,7 @@ per tiny thing is the accretion trap in a new costume.
61
61
  - steward: `ruler-hardening` (when/how to make the metric harder once it's gamed —
62
62
  add a transfer split, harder cells) · `benchmark-design` (noise floors, seeds,
63
63
  why a sweep beats a single point)
64
- - follow-up: `respond-to-review` (address maintainer comments; which task
64
+ - resumed author: `respond-to-review` (address maintainer comments; which task
65
65
  instruction a wake supersedes; honest scope)
66
66
 
67
67
  **Target (e.g. `yolo-jepa`)**
@@ -138,7 +138,7 @@ planned set — none are wired yet):
138
138
  | author | Read/Grep/Glob/Write/Edit/Bash | kernel-primer, plain-style, hypothesis-discipline, honest-method, experiment-lifecycle, research-report (+ self-review, analyze-results — proposed) | literature-search (retriever), result-aggregation |
139
139
  | reviewer | Read/Grep/Glob + pr-context-read, retriever | kernel-primer, plain-style, review-rubric, read-only-investigation | evidence-sweep (parallel read-only file/caller sweeps on large diffs), reference-check (retriever) |
140
140
  | verifier | Read/Grep/Glob + pr-context-read, retriever | kernel-primer, plain-style, integrity-lens, read-only-investigation | evidence-sweep (read-only; e.g. trace every consumer of a changed ruler input) |
141
- | followup | editing set, resuming role's key/scope | kernel-primer, plain-style, respond-to-review | inherits the resumed role's |
141
+ | resumed author | editing set, resuming role's key/scope | kernel-primer, plain-style, respond-to-review | inherits the resumed role's |
142
142
  | steward | editing set, own territory | kernel-primer, plain-style, ruler-hardening, benchmark-design | literature-search (eval conventions) |
143
143
 
144
144
  The self-review skill's relationship to the panel is specified in
@@ -252,7 +252,7 @@ is the rare exception).
252
252
 
253
253
  | Field | Meaning |
254
254
  | --- | --- |
255
- | `name` | role id (author, reviewer, verifier, steward, followup) |
255
+ | `name` | role id (author, reviewer, verifier, steward) |
256
256
  | `instructions` | standing role prompt, composed from skills |
257
257
  | `skills` | skill ids to load (global + role + target) |
258
258
  | `tools` | allowed tools (native + harness-provided) |
@@ -267,7 +267,7 @@ is the rare exception).
267
267
 
268
268
  Replaces the five per-role drivers' session dispatch — all five roles now run
269
269
  through `run_role` with their RoleSpec (`review_cli` and `verifier_cli` are
270
- deleted; author, follow-up, and steward dispatch from their kernel modules).
270
+ deleted; author and steward dispatch from their kernel modules).
271
271
  Kernel code; it calls into the agentic realm at step 2, and everything
272
272
  trust-critical is deterministic.
273
273
 
@@ -8,7 +8,8 @@ review.
8
8
  A background agent that co-develops the lab's benchmark-bearing repos (jepa-agent,
9
9
  egolearn): picks work from their roadmaps and benchmark gaps, implements on a
10
10
  branch, runs GPU experiments, opens a PR when a metric improves, reviews PRs, and
11
- reports weekly. Humans keep the merge button.
11
+ reports weekly. Humans keep the merge button unless a repo owner opts in to
12
+ `merge: auto` in the contract.
12
13
 
13
14
  ## Decisions
14
15
 
@@ -180,12 +181,12 @@ email, no command grammar to learn or to secure.
180
181
  its own.
181
182
 
182
183
  **In flight — humans steer through the PR.** A run whose PR is open enters
183
- `in-review` and stays alive: each tick, the sweep checks the PR for new
184
+ `parked` and stays alive: each tick, the sweep checks the PR for new
184
185
  comments by org members (the advisory reviewer never comments on bot PRs, so
185
186
  no bot-to-bot loop can form). A qualifying comment wakes the run — the same
186
187
  resume mechanism as experiment wakes, comment text data-fenced, task-level
187
- supersession only — and the agent pushes fixes and replies on the thread via
188
- the bot identity. Review-response sessions are bounded like everything else
188
+ supersession only. The author replies on the thread and submits changes
189
+ through the gate and publish. Comments wait while launched jobs are active. Review-response sessions are bounded like everything else
189
190
  (attempt counter, budget). Non-member comments never trigger a wake.
190
191
 
191
192
  **Death — a run ends in exactly one of six ways, each producing a report:**
@@ -209,7 +210,7 @@ honest wording, rather than waiting for the orphan-reconciliation pass to
209
210
  mislabel a deliberate "no" as a crash.
210
211
 
211
212
  After the report is distilled into `lessons/`, the workspace and per-run HOME
212
- are garbage-collected (grace period first — an `in-review` run's context must
213
+ are garbage-collected (grace period first — a `parked` run's context must
213
214
  survive until its PR closes). The notebook is the memory; the run directory
214
215
  is scaffolding.
215
216
 
@@ -363,11 +364,11 @@ failure can strand a run:
363
364
  strand the run.
364
365
  3. *Backup — the tick sweeps.* The tick chain (independently kept alive:
365
366
  two queued successors, heartbeat, GH-Actions watchdog) scans every run in
366
- `waiting` each tick: experiment job terminal per `sacct` + no wake lease
367
+ `parked` each tick: experiment job terminal per `sacct` + no wake lease
367
368
  or completion within a grace window → the tick dispatches the wake
368
369
  itself. This covers a lost wake job, a wake killed mid-session, and Slurm
369
370
  controller restarts that drop pending jobs.
370
- 4. *Deadline floor.* Every `waiting` run records
371
+ 4. *Deadline floor.* Every `parked` run records
371
372
  `deadline = submit_time + walltime + slack` (recomputed from `start_time`
372
373
  once the job starts, so late scheduling never truncates a healthy run).
373
374
  Past the deadline the sweep consults `sacct` and acts on what it finds:
@@ -537,5 +538,9 @@ not yet build-gating.
537
538
  - GitHub App (revisit if PAT limits bite)
538
539
  - Cloud compute backend (burst valve when Torch queues block)
539
540
  - Third-party / self-hosted models
540
- - Any form of auto-merge on code — never. The sole exception is the prose-only
541
- notebook repo.
541
+ - Auto-merge on code is off by default and is never the kernel's decision.
542
+ A repo owner may turn it on with `merge: auto` in the contract: a PR whose
543
+ gate and panel read are clean is merged by the kernel sweep, at the blessed
544
+ head only, through the repo's required checks and branch protection. The
545
+ kernel never arms GitHub auto-merge in this mode and withdraws old arms.
546
+ The prose-only notebook repo merges on its secret-scan check alone.