outerloop-science 0.1.0.dev1__tar.gz → 0.1.0.dev2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/CHANGELOG.md +67 -0
  2. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/PKG-INFO +2 -2
  3. outerloop_science-0.1.0.dev2/containers/README.md +8 -0
  4. outerloop_science-0.1.0.dev2/containers/agent-py312.def +43 -0
  5. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/architecture.md +6 -2
  6. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/dispatcher.md +17 -3
  7. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/github-app-auth.md +5 -5
  8. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/headline.md +1 -1
  9. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/resident-tick.md +5 -5
  10. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/role-cli.md +4 -4
  11. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/install.md +59 -6
  12. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/roadmap.md +1 -1
  13. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/validation/author-syscalls.md +5 -5
  14. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/pyproject.toml +6 -3
  15. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/README.md +1 -1
  16. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/install_codex.sh +2 -2
  17. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/tick_chain.sbatch +41 -35
  18. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/tick_deploy.sh +42 -38
  19. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/tick_resident.sh +23 -20
  20. outerloop_science-0.1.0.dev2/src/outerloop/__init__.py +18 -0
  21. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/appauth.py +3 -3
  22. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/appmanifest.py +1 -1
  23. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/attempt.py +55 -49
  24. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/cli.py +150 -74
  25. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/compute.py +59 -9
  26. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/contract.py +3 -2
  27. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/followup.py +14 -17
  28. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/github.py +4 -4
  29. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/harness.py +18 -7
  30. outerloop_science-0.1.0.dev2/src/outerloop/image.py +372 -0
  31. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/init.py +229 -14
  32. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/limits.py +1 -1
  33. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/measure.py +2 -2
  34. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/orchestrator.py +1 -1
  35. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/paths.py +2 -1
  36. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/review.py +1 -1
  37. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/steward.py +3 -3
  38. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/syscall.py +1 -0
  39. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/tick.py +302 -77
  40. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_attempt.py +20 -20
  41. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_bot_aliases.py +7 -7
  42. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_bot_login.py +5 -5
  43. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_compute.py +42 -12
  44. outerloop_science-0.1.0.dev2/tests/test_env_bridge.py +86 -0
  45. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_harness.py +35 -8
  46. outerloop_science-0.1.0.dev2/tests/test_image.py +294 -0
  47. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_init.py +295 -4
  48. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_local_compute.py +27 -6
  49. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_measure.py +1 -1
  50. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_start.py +169 -65
  51. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_tick.py +452 -89
  52. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_tick_chain_successors.py +2 -2
  53. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_tick_resident.py +33 -32
  54. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_verify_agent.py +13 -11
  55. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/uv.lock +2 -6
  56. outerloop_science-0.1.0.dev1/src/outerloop/__init__.py +0 -18
  57. outerloop_science-0.1.0.dev1/tests/test_env_bridge.py +0 -61
  58. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/.gitignore +0 -0
  59. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/.pre-commit-config.yaml +0 -0
  60. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/.python-version +0 -0
  61. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/CLAUDE.md +0 -0
  62. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/CONTRIBUTING.md +0 -0
  63. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/LICENSE +0 -0
  64. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/NOTICE +0 -0
  65. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/README.md +0 -0
  66. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/RELEASING.md +0 -0
  67. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/SECURITY.md +0 -0
  68. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/assets/icon-dark.svg +0 -0
  69. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/assets/icon-light.svg +0 -0
  70. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/assets/icon.svg +0 -0
  71. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/compute.md +0 -0
  72. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/contract.md +0 -0
  73. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/agent-substrate.md +0 -0
  74. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/consolidation.md +0 -0
  75. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/external.md +0 -0
  76. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/judge-placement.md +0 -0
  77. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/meta.md +0 -0
  78. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/onboarding.md +0 -0
  79. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/orchestrator-verify.md +0 -0
  80. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/public-surface.md +0 -0
  81. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/research-lines.md +0 -0
  82. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/research-loop-buildout.md +0 -0
  83. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/research-loop.md +0 -0
  84. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/review-placement.md +0 -0
  85. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/reviewer-infra.md +0 -0
  86. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/roles.md +0 -0
  87. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/design/scaling.md +0 -0
  88. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/docs/reviewer.md +0 -0
  89. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/examples/review.yml +0 -0
  90. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/install_hermes.sh +0 -0
  91. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/requeue_moved_successors.sh +0 -0
  92. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/setup_branch_protection.sh +0 -0
  93. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/scripts/sweep_git_locks.sh +0 -0
  94. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/__main__.py +0 -0
  95. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/brief.py +0 -0
  96. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/climbboard.py +0 -0
  97. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/contract_cli.py +0 -0
  98. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/disk.py +0 -0
  99. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/dispatch.py +0 -0
  100. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/housekeeping.py +0 -0
  101. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/intake.py +0 -0
  102. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/markers.py +0 -0
  103. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/panel.py +0 -0
  104. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/posting.py +0 -0
  105. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/progress.py +0 -0
  106. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/py.typed +0 -0
  107. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/review_agent.py +0 -0
  108. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/review_agent_cli.py +0 -0
  109. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/review_post_cli.py +0 -0
  110. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/review_summarize_cli.py +0 -0
  111. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/role_runner.py +0 -0
  112. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/roles.py +0 -0
  113. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/rolespec.py +0 -0
  114. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/runstate.py +0 -0
  115. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/style.py +0 -0
  116. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/syscall_cli.py +0 -0
  117. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/verifier.py +0 -0
  118. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/verify_agent.py +0 -0
  119. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/verify_agent_cli.py +0 -0
  120. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/src/outerloop/verify_post_cli.py +0 -0
  121. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/conftest.py +0 -0
  122. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_appauth.py +0 -0
  123. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_appmanifest.py +0 -0
  124. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_brief.py +0 -0
  125. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_channel_dir.py +0 -0
  126. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_climbboard.py +0 -0
  127. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_codex_harness.py +0 -0
  128. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_contract.py +0 -0
  129. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_contract_cli.py +0 -0
  130. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_contract_names.py +0 -0
  131. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_disk.py +0 -0
  132. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_dispatch.py +0 -0
  133. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_followup.py +0 -0
  134. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_github.py +0 -0
  135. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_hardening.py +0 -0
  136. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_hermes_harness.py +0 -0
  137. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_housekeeping.py +0 -0
  138. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_import.py +0 -0
  139. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_intake.py +0 -0
  140. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_limits.py +0 -0
  141. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_markers.py +0 -0
  142. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_measure_and_decide.py +0 -0
  143. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_orchestrator.py +0 -0
  144. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_packaging.py +0 -0
  145. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_panel.py +0 -0
  146. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_paths.py +0 -0
  147. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_posting.py +0 -0
  148. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_progress.py +0 -0
  149. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_requeue_moved_successors.py +0 -0
  150. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review.py +0 -0
  151. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review_agent.py +0 -0
  152. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review_agent_cli.py +0 -0
  153. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review_hardening.py +0 -0
  154. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review_policy.py +0 -0
  155. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_review_summarize.py +0 -0
  156. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_role_runner.py +0 -0
  157. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_rolespec.py +0 -0
  158. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_runstate.py +0 -0
  159. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_steward.py +0 -0
  160. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_sweep_git_locks.py +0 -0
  161. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_syscall.py +0 -0
  162. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_syscall_cli.py +0 -0
  163. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_tiers.py +0 -0
  164. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_verifier.py +0 -0
  165. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_verify_agent_cli.py +0 -0
  166. {outerloop_science-0.1.0.dev1 → outerloop_science-0.1.0.dev2}/tests/test_version.py +0 -0
@@ -21,6 +21,25 @@ Versions follow [SemVer](https://semver.org).
21
21
  before the rename. The focused `--github-app` run still asks nothing about
22
22
  the author. Every secret `init` writes (keys, PAT, App PEM and JSON, `.env`)
23
23
  is now created 0600 in one step rather than written and then tightened.
24
+ - The agent image is published: `containers/agent-py312.def` is its recipe
25
+ (Ubuntu 24.04, Python 3.12, uv, git, build tools), the `build-image` workflow
26
+ builds and uploads it with a checksum to huggingface.co/outerloop-science/
27
+ agent-image on every recipe change, and `outerloop init` downloads it on
28
+ Linux when Apptainer is installed, so local mode runs contained by default
29
+ (`--image` names your own, `--no-image` opts out).
30
+ - `outerloop init` checks that Apptainer can actually run a container before
31
+ downloading the image (on Ubuntu 24.04 a hand-installed Apptainer is on PATH
32
+ but cannot create user namespaces), and when it cannot, prints the install
33
+ steps for the machine's distribution and continues uncontained. The image
34
+ download shows a progress bar with size, speed and ETA on a terminal, one
35
+ line per 10% otherwise, and confirms the checksum. `docs/install.md` gains
36
+ an "Installing Apptainer" section.
37
+ - Launch admission: with `OUTERLOOP_MAX_LAUNCH_GPUS` set to the per-user GPU
38
+ cap, author launches are submitted held and the tick releases them
39
+ oldest-first while the user's GPU jobs fit under the cap, cancels held
40
+ launches of runs that have ended, and stops releasing when Slurm parks a
41
+ released launch on a per-user reason. `squeue` rows on the board carry the
42
+ pending reason and GRES. Unset, launches queue as before.
24
43
 
25
44
  ### Changed
26
45
 
@@ -43,6 +62,25 @@ Versions follow [SemVer](https://semver.org).
43
62
  (a `-m` in `addopts` disables testmon), and a `serial` tier holds the tests
44
63
  that inspect the process table: skipped while workers run, and `pytest -m
45
64
  serial` turns workers off, so the tier always runs alone (CI's second step).
65
+ - The kernel reads `OUTERLOOP_*` everywhere: every internal read, log line,
66
+ chain script and test now uses the public names. The pre-rename
67
+ `AUTORESEARCH_*` names are still accepted for one release, bridged at the
68
+ process boundary, the `.env` file and the chain scripts, and will be
69
+ dropped in the release after 0.1. The review footer names `outerloop`, and
70
+ `outerloop --version` exists; `start`, `tick` and `init` describe themselves
71
+ and every flag in `--help`.
72
+ - The on-disk and queue names follow the rename: the resident tick job is
73
+ `outerloop-resident` (the per-cadence chain `outerloop-tick`), the local
74
+ state root defaults to `~/.outerloop`, the default image path is
75
+ `~/outerloop-images/agent-py312.sif`. `outerloop start` refuses while a
76
+ pre-rename `autoresearch-resident` is queued or running, since Slurm's
77
+ singleton serializes by name; cancel it first. An existing `~/.autoresearch`
78
+ or `~/autoresearch-images/` is used while the new path does not exist. A
79
+ chain-mode deployment (manual `tick_chain.sbatch`) must be stopped with
80
+ `scancel --name autoresearch-tick` before deploying this version. Every
81
+ honored pre-rename name (`~/.config/autoresearch/`, `.autoresearch.yaml`, the
82
+ `.autoresearch` channel dir, the old root, image and job names) is dropped in
83
+ the release after 0.1; the comment marker keeps recognizing old comments.
46
84
 
47
85
  ### Changed
48
86
 
@@ -128,6 +166,35 @@ Versions follow [SemVer](https://semver.org).
128
166
  - `outerloop init` no longer overwrites an existing `~/.config/autoresearch/.env`
129
167
  silently: interactively it asks; with `--yes` it refuses unless `--force` is
130
168
  passed. The check runs before any GitHub App is created.
169
+ - Slurm account and partition are both optional (#300). `outerloop start`
170
+ no longer requires an account, `outerloop init` no longer insists on one,
171
+ the tick's in-review servicing and the attempt's resume and dispatch gates
172
+ need only the image, and every sbatch the kernel builds passes `--account`
173
+ and `--partition` only when set, so Slurm bills the caller's default
174
+ association and picks its default partition.
175
+ - A launch still queued past its deadline is no longer cancelled and woken as
176
+ "unschedulable" when Slurm's pending reason is a wait: a busy queue
177
+ (Priority, Resources), a reservation or dependency, or the account's own
178
+ per-user/per-account/group cap (`QOSMaxGRESPerUser` and kin). The sweep
179
+ extends the deadline by one slack window and logs the reasons; a queued job
180
+ keeps its priority age, where a re-launch would start over at the back.
181
+ Reasons that never clear (a dependency that cannot be satisfied, per-job
182
+ limits, an invalid account or QOS, a held job) still cancel and wake.
183
+ - `outerloop init` stops with exit 1 when the credential cannot open pull
184
+ requests on the target (401, 403, 404, or no write access) instead of
185
+ printing a warning and `next: outerloop start` (#285); a network failure stays
186
+ a warning. The PAT path records the token's login as `OUTERLOOP_BOT_LOGIN`,
187
+ and the tick warns when the login is unset and the lab default applies (#298).
188
+ - `init` locates the author's CLI (`claude`/`codex`) on PATH or in
189
+ `~/.local/bin`, records it as `OUTERLOOP_CLAUDE_BIN`/`OUTERLOOP_CODEX_BIN`, and
190
+ says how to install it when absent; `attempt --claude-bin` honors that
191
+ setting; `start` refuses to launch when a recorded binary is missing; a
192
+ `spawn-error` names the binary and the OS error (#294).
193
+ - Local jobs keep their combined stdout/stderr in `local_jobs/<id>.out` beside
194
+ the state file, and the log line names it when the job did not complete (#295).
195
+ - `cryptography` is a base dependency, so the GitHub App identity works from a
196
+ plain `pip install`; the `app-auth` extra is now a no-op (#293). The `tick`
197
+ subcommand's help no longer uses lab-internal vocabulary (#286).
131
198
 
132
199
  ### Added
133
200
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: outerloop-science
3
- Version: 0.1.0.dev1
3
+ Version: 0.1.0.dev2
4
4
  Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
5
5
  Project-URL: Homepage, https://outerloop.science
6
6
  Project-URL: Repository, https://github.com/outerloop-science/outerloop
@@ -11,10 +11,10 @@ License-Expression: Apache-2.0
11
11
  License-File: LICENSE
12
12
  License-File: NOTICE
13
13
  Requires-Python: >=3.12
14
+ Requires-Dist: cryptography>=42
14
15
  Requires-Dist: pydantic>=2
15
16
  Requires-Dist: pyyaml>=6
16
17
  Provides-Extra: app-auth
17
- Requires-Dist: cryptography>=42; extra == 'app-auth'
18
18
  Description-Content-Type: text/markdown
19
19
 
20
20
  <picture>
@@ -0,0 +1,8 @@
1
+ # containers
2
+
3
+ `agent-py312.def` is the Apptainer recipe for the image the kernel runs sessions and
4
+ evaluations in. The `build-image` workflow builds it on every change to this directory and
5
+ on manual dispatch, and publishes the result to
6
+ [huggingface.co/outerloop-science/agent-image](https://huggingface.co/outerloop-science/agent-image)
7
+ with a checksum. `outerloop init` downloads it on Linux when Apptainer is installed; the
8
+ tick reads `~/outerloop-images/agent-py312.sif` unless `OUTERLOOP_IMAGE` says otherwise.
@@ -0,0 +1,43 @@
1
+ # The Outerloop agent image: what a contained session and a dispatched
2
+ # evaluation see. The kernel binds the workspace, a per-run home and the
3
+ # harness binary (claude/codex) into it; --nv adds the host's GPU driver for
4
+ # GPU benchmarks. So the image only has to provide the base a target's own
5
+ # `uv run ...` needs: Python 3.12, uv, git and a compiler.
6
+ #
7
+ # Built by .github/workflows/build-image.yml and published to
8
+ # https://huggingface.co/outerloop-science/agent-image; `outerloop init` pulls
9
+ # it on Linux. Build by hand: apptainer build agent-py312.sif containers/agent-py312.def
10
+
11
+ Bootstrap: docker
12
+ From: ubuntu:24.04
13
+
14
+ %post
15
+ set -eu
16
+ export DEBIAN_FRONTEND=noninteractive
17
+ apt-get update
18
+ apt-get install -y --no-install-recommends \
19
+ ca-certificates curl git git-lfs openssh-client \
20
+ build-essential pkg-config \
21
+ python3 python3-venv python3-dev \
22
+ libgomp1 libglib2.0-0 procps less unzip zip jq ripgrep
23
+ rm -rf /var/lib/apt/lists/*
24
+ # uv installs its own Pythons when a target pins one; the system 3.12 is the default
25
+ curl -LsSf https://astral.sh/uv/install.sh | env UV_INSTALL_DIR=/usr/local/bin INSTALLER_NO_MODIFY_PATH=1 sh
26
+ ln -sf /usr/bin/python3 /usr/local/bin/python
27
+ git lfs install --system
28
+ python3 --version && uv --version && git --version
29
+
30
+ %environment
31
+ export LC_ALL=C.UTF-8
32
+ export LANG=C.UTF-8
33
+ export PATH=/usr/local/bin:/usr/local/sbin:/usr/sbin:/usr/bin:/sbin:/bin
34
+
35
+ %labels
36
+ org.opencontainers.image.title outerloop agent image
37
+ org.opencontainers.image.source https://github.com/outerloop-science/outerloop
38
+ outerloop.recipe containers/agent-py312.def
39
+
40
+ %help
41
+ Outerloop agent image: Ubuntu 24.04, Python 3.12, uv, git, build tools.
42
+ The kernel runs sessions and evaluations in it with --containall --cleanenv,
43
+ binding only the workspace, a per-run home and the harness binary.
@@ -371,8 +371,12 @@ failure can strand a run:
371
371
  `deadline = submit_time + walltime + slack` (recomputed from `start_time`
372
372
  once the job starts, so late scheduling never truncates a healthy run).
373
373
  Past the deadline the sweep consults `sacct` and acts on what it finds:
374
- still PENDING → the experiment is unschedulable in practice; cancel it and
375
- wake the run with that fact. No record on a *successful* query → the job
374
+ still PENDING → ask squeue why: a wait (Priority/Resources, a reservation
375
+ or dependency, the account's own per-user/group cap such as
376
+ `QOSMaxGRESPerUser`) moves the deadline out by one slack window, since a
377
+ queued job keeps its priority age and a re-launch would start over; any
378
+ other reason means unschedulable in practice — cancel it and wake the
379
+ run with that fact. No record on a *successful* query → the job
376
380
  vanished; wake with that fact. Query failed or timed out → that is
377
381
  "Slurm unknown", never "job gone" — defer to the next tick, and only a
378
382
  sustained outage (its own alert via the watchdog) escalates. A fail-safe
@@ -66,7 +66,7 @@ orchestrator chooses per eval site, not per repo.
66
66
  launches alike: the JobSpec requests `--gpus=N`, and the jail adds `--nv` so
67
67
  the allocation is visible inside the container; nothing else about the
68
68
  containment changes. GPU jobs need their own lane, so the deployment names
69
- one — `AUTORESEARCH_GPU_PARTITION`, optionally `AUTORESEARCH_GPU_ACCOUNT`
69
+ one — `OUTERLOOP_GPU_PARTITION`, optionally `OUTERLOOP_GPU_ACCOUNT`
70
70
  (default: the CPU account) — and a benchmark with `gpus > 0` on a deployment
71
71
  without a lane is REFUSED at launch (both attempt lanes), never queued into
72
72
  evals that can never run. GPUs exist only on DISPATCHED jobs: the contract
@@ -76,6 +76,18 @@ benchmarks outright for now — a stewardship validates its rewrite in-job.
76
76
  The first target in this shape is the speedrun (`gpt-speedrun`: one H200
77
77
  per eval, ~3.5h).
78
78
 
79
+ **Launch admission.** A per-user GPU cap (a QOS's `MaxTRESPerUser`, 16 on
80
+ Torch) means the fleet can run only so many launches at once, whatever the
81
+ agent count: six 8-GPU launches queued against a 16-GPU cap left four pending
82
+ for a day on `QOSMaxGRESPerUser`. With `OUTERLOOP_MAX_LAUNCH_GPUS` set to that
83
+ cap, author launches are submitted HELD, and every tick `service_admission`
84
+ releases them oldest-first while the user's eligible GPU jobs (evals and
85
+ launches alike) fit under the cap; held launches whose run has since ended
86
+ are cancelled. A released launch that Slurm parks on a per-user reason means
87
+ the cap is set too high, and nothing more is released that tick. The wake's
88
+ `afterany` dependency is unchanged (held jobs are pending), and the sweep
89
+ treats the kernel's hold as a wait. Unset, launches queue as submitted.
90
+
79
91
  **Who decides the eval walltime.** The gate measures steps, not time, so
80
92
  walltime should bound only SPEND — and the attempt already has a spend
81
93
  budget. The contract's `eval_minutes` is the DEFAULT; the author may declare
@@ -234,7 +246,9 @@ dies -> the `eval-error` outcome at wake (ending `aborted`, exactly as the
234
246
  in-job eval failure maps today); a LOST wake -> the sweep's primary backup,
235
247
  which wakes any run whose job reads terminal after the grace period —
236
248
  plus GONE past the deadline (vanished) and PENDING past the deadline
237
- (cancel, then wake as unschedulable); a still-RUNNING job is deliberately
249
+ (when Slurm's reason is a wait — a busy queue, a reservation or dependency,
250
+ the account's own per-user/group cap — the deadline moves out by one slack
251
+ window; otherwise cancel, then wake as unschedulable); a still-RUNNING job is deliberately
238
252
  left alone, bounded by its own walltime; wakes that fire without producing
239
253
  progress -> `stuck` at MAX_WAKE_ATTEMPTS (a landed re-measure has its own
240
254
  allowance of MAX_WAKE_ATTEMPTS finishing follow-ups, counted on the stage). A moved base during the wait is
@@ -287,7 +301,7 @@ GPU benchmarks and portfolio climbs both multiply eval load; both were
287
301
  designed against this seam. Portfolio's N concurrent climbs become N
288
302
  waiting records with independent wakes — the serialization question stays
289
303
  in the picker, not the dispatcher. The tick's job-partition knobs
290
- (`AUTORESEARCH_JOB_PARTITION`) already route work jobs; per-benchmark
304
+ (`OUTERLOOP_JOB_PARTITION`) already route work jobs; per-benchmark
291
305
  partition/GPU allocation fields are a phase-3 contract-schema change
292
306
  (`Benchmark` rejects unknown fields today, deliberately), validated and
293
307
  clamped by operator ceilings like every other budget.
@@ -108,7 +108,7 @@ Swapping the live fleet's identity mid-campaign is an ops event:
108
108
 
109
109
  ## The flag
110
110
 
111
- `AUTORESEARCH_GITHUB_APP_FILE` names a JSON config —
111
+ `OUTERLOOP_GITHUB_APP_FILE` names a JSON config —
112
112
 
113
113
  ```json
114
114
  {"app_id": 1234, "installation_id": 5678,
@@ -117,13 +117,13 @@ Swapping the live fleet's identity mid-campaign is an ops event:
117
117
 
118
118
  — and rides the rails the PAT already rides: role CLIs default their
119
119
  `--github-app-file` from it (jobs inherit the tick's environment, the same
120
- way `AUTORESEARCH_AUTHOR_*` reaches them), and every bot-auth construction
120
+ way `OUTERLOOP_AUTHOR_*` reaches them), and every bot-auth construction
121
121
  goes through one factory, `appauth.resolve_bot_auth(pat_file, app_file)`:
122
122
  App provider when the config is set, PAT otherwise. Revert = unset the env
123
123
  var. The ids are not secrets; the key path inside keeps PAT-file custody
124
124
  (600, never committed).
125
125
 
126
- The identity flips WITH the credential: `AUTORESEARCH_BOT_LOGIN` names the
126
+ The identity flips WITH the credential: `OUTERLOOP_BOT_LOGIN` names the
127
127
  login the kernel now posts as (`<app-slug>[bot]`, e.g.
128
128
  `outerloop-autoresearch[bot]`), read by every role through one
129
129
  `github.bot_login_from_env()` default. Every own-comment filter, alarm-issue
@@ -135,10 +135,10 @@ Everything the kernel created BEFORE the flip carries the old login — the
135
135
  research-log issue, contract alarms, intake claims, its PRs. Recognition is
136
136
  keyed on login on purpose (the markers are public strings anyone can paste),
137
137
  so a flip must widen the set of our logins, not loosen the check:
138
- `AUTORESEARCH_BOT_ALIASES=agentic-learning-bot` names the former identity,
138
+ `OUTERLOOP_BOT_ALIASES=agentic-learning-bot` names the former identity,
139
139
  and every "is this ours" gate goes through `github.is_own_login`. The
140
140
  reusable review and verify workflows take the same value as a `bot_aliases`
141
- input (exported as `AUTORESEARCH_BOT_ALIASES`; the verifier's author gate
141
+ input (exported as `OUTERLOOP_BOT_ALIASES`; the verifier's author gate
142
142
  accepts an alias too), so a target repo passes `bot_login` = the App login and
143
143
  `bot_aliases` = the former account. Live lesson: without it, the first tick
144
144
  under the App claimed the kernel's own research-log issue as a research order.
@@ -61,7 +61,7 @@ monitoring agent + ~100 interventions; AIDE: in-process tree search). Ours is
61
61
  **decentralized on Slurm**: no resident daemon — a self-perpetuating chain of
62
62
  ~16 s stateless ticks; all state in typed records on the shared FS; every
63
63
  role (author session, eval, panel judge, wake) its own Slurm job — wake
64
- dispatch sits behind an operator flag (`AUTORESEARCH_DISPATCH_WAKE`, on in
64
+ dispatch sits behind an operator flag (`OUTERLOOP_DISPATCH_WAKE`, on in
65
65
  our deployment since 2026-08-20; retiring the dark-launch flag to
66
66
  default-on is a listed cleanup); crash or preemption anywhere heals
67
67
  through the sweep into honest endings.
@@ -1,7 +1,7 @@
1
1
  # The resident tick
2
2
 
3
3
  *Design note, 2026-09-02. Status: built as the opt-in mode
4
- (`AUTORESEARCH_RESIDENT=1`, `scripts/tick_resident.sh`); the chain restarts of
4
+ (`OUTERLOOP_RESIDENT=1`, `scripts/tick_resident.sh`); the chain restarts of
5
5
  2026-09-02 are the evidence.*
6
6
 
7
7
  ## The problem
@@ -91,16 +91,16 @@ covers it.
91
91
 
92
92
  ## Rollout
93
93
 
94
- 1. Opt-in mode in `scripts/tick_chain.sbatch`: `AUTORESEARCH_RESIDENT=1` in
94
+ 1. Opt-in mode in `scripts/tick_chain.sbatch`: `OUTERLOOP_RESIDENT=1` in
95
95
  the chain's environment selects the loop; unset keeps the per-cadence
96
96
  chain. The walltime is fixed by Slurm before the script runs, so the
97
97
  resident chain is STARTED with an explicit walltime and its own job name:
98
98
 
99
99
  ```
100
- sbatch --time=360 --job-name=autoresearch-resident --dependency=singleton \
100
+ sbatch --time=360 --job-name=outerloop-resident --dependency=singleton \
101
101
  --account=… --partition=cpu_short \
102
- --export=ALL,AUTORESEARCH_RESIDENT=1,AUTORESEARCH_HOME=…,AUTORESEARCH_ROOT=…,\
103
- AUTORESEARCH_ACCOUNT=…,AUTORESEARCH_PARTITION=cpu_short,AUTORESEARCH_PAT_FILE=… \
102
+ --export=ALL,OUTERLOOP_RESIDENT=1,OUTERLOOP_HOME=…,OUTERLOOP_ROOT=…,\
103
+ OUTERLOOP_ACCOUNT=…,OUTERLOOP_PARTITION=cpu_short,OUTERLOOP_PAT_FILE=… \
104
104
  scripts/tick_chain.sbatch
105
105
  ```
106
106
 
@@ -40,7 +40,7 @@ The invariants, proven on #132/#133 and non-negotiable everywhere:
40
40
  in the same PR — never two ways to say the same thing.
41
41
  4. **Verbs are RoleSpec-gated.** The RoleSpec already caps each role's
42
42
  tools/scope/key; the CLI's live verbs are part of that cap. It is ONE tool
43
- (`.autoresearch/syscall`) whose verbs the brief exposes per role — an author
43
+ (`.outerloop/syscall`) whose verbs the brief exposes per role — an author
44
44
  uses `launch`/`sleep`, a judge uses `finding`/`conclude` — and a syscall is
45
45
  TYPED, so the kernel dispatches by type (a sleep parks + wakes; a verdict is
46
46
  read back). Every role runs the tool the same way: a shell in the jail. Roles
@@ -49,7 +49,7 @@ The invariants, proven on #132/#133 and non-negotiable everywhere:
49
49
 
50
50
  ## End-state
51
51
 
52
- One `.autoresearch/syscall` tool, its live verbs gated per role by RoleSpec:
52
+ One `.outerloop/syscall` tool, its live verbs gated per role by RoleSpec:
53
53
 
54
54
  | Role | Verbs | Win | Installed where |
55
55
  | --- | --- | --- | --- |
@@ -70,7 +70,7 @@ lands with tests, and deletes what it replaces.
70
70
  ### Phase 0 — finish Phase A (in flight, prerequisite)
71
71
 
72
72
  The wake for `author-sleep` parks: gather each launch's results from the run
73
- dir, deliver declared artifacts into `.autoresearch/results/<name>/`, resume
73
+ dir, deliver declared artifacts into `.outerloop/results/<name>/`, resume
74
74
  the **same session** through the climb's resume-entry (#129) with the results
75
75
  data-fenced, update the budget file. (The transitional `AUTHOR_SLEEP_WAKE_READY`
76
76
  constant and env-flag arming have since retired: enablement is contract-driven.) Plus
@@ -95,7 +95,7 @@ findings drafts the PR for a human. A submit costs the sleep it rides on.
95
95
 
96
96
  A verdict is not a second tool — it is a syscall TYPE on the one surface.
97
97
  Reviewer/verifier/panel stop emitting one schema-constrained final message.
98
- Instead: `python .autoresearch/syscall finding --file --line --confidence
98
+ Instead: `python .outerloop/syscall finding --file --line --confidence
99
99
  --summary --detail [--blocking] [--kind] [--category]` per finding, then
100
100
  `... conclude --notes ...` to commit — each call validated on the spot, the
101
101
  verdict assembled kernel-side (`type: "verdict"`), well-formed by construction.
@@ -214,14 +214,19 @@ backend; the interface is small (submit a job, poll for completion), so a CI
214
214
  runner, a cloud backend, or a hardware rig plugs in the same way.
215
215
 
216
216
  The deployment is configured by environment (`OUTERLOOP_*`; the pre-rename
217
- `AUTORESEARCH_*` names are still accepted everywhere). Placement and paths are set
217
+ `AUTORESEARCH_*` names are still accepted for one release). Placement and paths are set
218
218
  when the chain is started: `OUTERLOOP_ACCOUNT`/`OUTERLOOP_PARTITION`
219
- place the CPU jobs (ticks, author sessions), `OUTERLOOP_HOME`/
219
+ place the CPU jobs (ticks, author sessions; both are optional, unset lets
220
+ Slurm bill the default association and pick the default partition),
221
+ `OUTERLOOP_HOME`/
220
222
  `OUTERLOOP_ROOT` locate the checkout and the state, `OUTERLOOP_IMAGE`
221
223
  the container. The rest is re-read from `~/.config/outerloop/.env` each
222
224
  tick, so changes take effect at the next cadence: `OUTERLOOP_TARGET`
223
225
  names the repo being climbed; `OUTERLOOP_GPU_PARTITION` (optionally
224
- `OUTERLOOP_GPU_ACCOUNT`) is the lane for GPU evals and launches — a
226
+ `OUTERLOOP_GPU_ACCOUNT`) is the lane for GPU evals and launches;
227
+ `OUTERLOOP_MAX_LAUNCH_GPUS` is the per-user GPU cap (your QOS's
228
+ `MaxTRESPerUser`) under which the tick admits author launches, unset = queue
229
+ them as submitted — a
225
230
  comma-separated partition list lets Slurm start each job wherever it fits
226
231
  first; `OUTERLOOP_PANEL` names the verify/review lenses (with
227
232
  `OUTERLOOP_PANEL_*_KEY_FILE` for their keys); the author backend is
@@ -232,7 +237,7 @@ first; `OUTERLOOP_PANEL` names the verify/review lenses (with
232
237
  still read). A Codex author always runs contained, so it also needs the image
233
238
  (`OUTERLOOP_IMAGE`) and a Codex model in `OUTERLOOP_AUTHOR_MODEL`. On a
234
239
  cluster, evals run inside the Apptainer image at `OUTERLOOP_IMAGE` (default
235
- `~/autoresearch-images/agent-py312.sif`) in a jail that binds only the
240
+ `~/outerloop-images/agent-py312.sif`) in a jail that binds only the
236
241
  checked-out tree — an eval that needs data must fetch it into the tree, and
237
242
  GPU jobs are requested per node (`--gpus-per-node`).
238
243
 
@@ -243,8 +248,56 @@ environment, both on your machine with your keys, and the loop says so once
243
248
  at start. The verification panel is off in this mode unless you set
244
249
  `OUTERLOOP_PANEL_UNCONTAINED=1`, because an uncontained judge holds a shell
245
250
  next to its own key file; a pull request opened without a panel says so. A
246
- Codex author needs the image in every mode. Contained local mode, the
247
- default on Linux once the image is published, needs Apptainer and the image.
251
+ Codex author needs the image in every mode. Contained local mode needs
252
+ Apptainer and the image: the published one lives at
253
+ [huggingface.co/outerloop-science/agent-image](https://huggingface.co/outerloop-science/agent-image),
254
+ built from `containers/agent-py312.def`. On Linux with Apptainer installed,
255
+ `outerloop init` downloads it to `~/outerloop-images/` (about 200 MB, with a
256
+ progress bar), verifies its published checksum and records it; `--image` points at
257
+ your own, `--no-image` keeps runs uncontained even when an image is already on disk
258
+ (it writes `OUTERLOOP_IMAGE=`, the off-switch). When Apptainer is missing or cannot
259
+ run containers, init says so, prints the install steps for your system, and
260
+ continues uncontained; run `outerloop init --force` after installing it.
261
+
262
+ **Installing Apptainer (Linux, one time, needs root or an admin).** Apptainer runs
263
+ sessions and evaluations in containers. Check for it
264
+ with `apptainer exec docker://alpine:3.20 cat /etc/alpine-release`; a version number
265
+ means it works.
266
+
267
+ - *Ubuntu.* Install from the project's PPA, which builds for amd64 and arm64. The
268
+ package carries the AppArmor profile Ubuntu 23.10 and later require; the
269
+ unprivileged installer does not, and fails at run time with
270
+ `Could not write info to setgroups`.
271
+
272
+ ```bash
273
+ sudo add-apt-repository -y ppa:apptainer/ppa
274
+ sudo apt-get update && sudo apt-get install -y apptainer
275
+ ```
276
+
277
+ - *Debian (x86-64).* Install the release package from
278
+ [github.com/apptainer/apptainer/releases](https://github.com/apptainer/apptainer/releases)
279
+ (`apptainer_<version>_amd64.deb`, not the `-suid` one):
280
+
281
+ ```bash
282
+ curl -fsSLO https://github.com/apptainer/apptainer/releases/download/v1.5.3/apptainer_1.5.3_amd64.deb
283
+ sudo apt-get install -y ./apptainer_1.5.3_amd64.deb
284
+ ```
285
+
286
+ The release has no package for other architectures; on ARM Debian, build from
287
+ source or use the unprivileged installer below.
288
+
289
+ - *Fedora.* `sudo dnf install -y apptainer`.
290
+ - *RHEL, Rocky, Alma.* `sudo dnf install -y epel-release`, then `sudo dnf install -y apptainer`.
291
+ - *No root.* The project's unprivileged installer works on most other systems. It is
292
+ pinned to the release tag; download it, read it, then run it:
293
+
294
+ ```bash
295
+ curl -fsSLO https://raw.githubusercontent.com/apptainer/apptainer/v1.5.3/tools/install-unprivileged.sh
296
+ less install-unprivileged.sh && bash install-unprivileged.sh ~/apptainer
297
+ export PATH=$HOME/apptainer/bin:$PATH
298
+ ```
299
+ - *Slurm clusters.* Ask the administrators; most already provide it.
300
+ - *macOS.* No Apptainer; runs stay uncontained (a macOS containment is on the roadmap).
248
301
 
249
302
  ---
250
303
 
@@ -238,7 +238,7 @@ Research findings sync through GitHub; job state never leaves its cluster.
238
238
  re-measured and re-gated); capped-out blocking opens a DRAFT PR and
239
239
  never arms auto-merge; the transcript rides in the PR body. DEPLOYED
240
240
  in code: the tick passes --panel verify,review to every climb job by
241
- default (off-switch AUTORESEARCH_PANEL=""). Cluster prerequisite: the
241
+ default (off-switch OUTERLOOP_PANEL=""). Cluster prerequisite: the
242
242
  verifier key file at ~/.config/outerloop/verifier_key on the tick
243
243
  account — a missing key fails climbs LOUDLY by design. GitHub-side
244
244
  verify.yml thins per target after the pilot runs clean
@@ -4,7 +4,7 @@
4
4
  > milestones green — the codex author launched + slept unprompted, the same
5
5
  > session resumed with its launch's post-hibernation results, and the run
6
6
  > reached an honest negative terminal. The `--author-syscalls` /
7
- > `AUTORESEARCH_AUTHOR_SYSCALLS` arming described by the original runbook has
7
+ > `OUTERLOOP_AUTHOR_SYSCALLS` arming described by the original runbook has
8
8
  > since RETIRED: enablement is contract-driven (dispatch coords + a resumable
9
9
  > backend + `depth_k > 0`; `depth_k: 0` is a benchmark's opt-out). The steps
10
10
  > below are kept for re-validation after substrate changes.
@@ -36,17 +36,17 @@ the brief only when armed).
36
36
 
37
37
  The lifecycle to confirm, in order:
38
38
 
39
- 1. **Tool installed:** `<run_dir>/ws/.autoresearch/syscall` exists, and
40
- `.autoresearch/budget.json` shows `depth_k` / `sleep_k`.
39
+ 1. **Tool installed:** `<run_dir>/ws/.outerloop/syscall` exists, and
40
+ `.outerloop/budget.json` shows `depth_k` / `sleep_k`.
41
41
  2. **Author sleeps:** the session ends having written
42
- `.autoresearch/syscall.json` (`type: "sleep"`, one+ launches).
42
+ `.outerloop/syscall.json` (`type: "sleep"`, one+ launches).
43
43
  3. **Park:** the run record is `waiting`, `stage.phase == "author-sleep"`, with
44
44
  `stage.afterany` naming the submitted launch job(s), `syscall_launches`, and
45
45
  `launches_used` / `sleeps_used` counts.
46
46
  4. **Launch job runs:** `<run_dir>/eval-launch-<name>/` fills with `exit-code`,
47
47
  `stdout`, `stderr`, and `artifacts/` (the declared files, copied out).
48
48
  5. **Wake:** the tick wakes the parked run; `gather_results` delivers artifacts
49
- into `<ws>/.autoresearch/results/<name>/`, and the SAME session resumes
49
+ into `<ws>/.outerloop/results/<name>/`, and the SAME session resumes
50
50
  (`resume_session_id` unchanged) with the results data-fenced + the author's
51
51
  note echoed back.
52
52
  6. **Terminate:** the woken author either sleeps again (a fresh author-sleep
@@ -12,6 +12,7 @@ license-files = ["LICENSE", "NOTICE"]
12
12
  requires-python = ">=3.12"
13
13
  authors = [{ name = "Agentic Learning AI Lab, New York University" }]
14
14
  dependencies = [
15
+ "cryptography>=42", # GitHub App auth (RS256 JWTs), the recommended identity
15
16
  "pydantic>=2",
16
17
  "pyyaml>=6",
17
18
  ]
@@ -24,9 +25,10 @@ dependencies = [
24
25
  outerloop = "outerloop.cli:main"
25
26
 
26
27
  [project.optional-dependencies]
27
- # GitHub App auth: RS256 signing for App JWTs (docs/design/github-app-auth.md).
28
- # Install where AUTORESEARCH_GITHUB_APP_FILE is set; the PAT path needs nothing.
29
- app-auth = ["cryptography>=42"]
28
+ # `cryptography` is a base dependency since 0.1.0.dev2 (the App identity init
29
+ # recommends must work from a plain install, #293); the extra stays as a no-op so
30
+ # `pip install "outerloop-science[app-auth]"` keeps working.
31
+ app-auth = []
30
32
 
31
33
  [dependency-groups]
32
34
  dev = [
@@ -75,6 +77,7 @@ testpaths = ["tests"]
75
77
  # the tests affected by your edits.
76
78
  addopts = "-ra --strict-markers -n auto"
77
79
  markers = [
80
+ "real_locate_harness: exercises the real harness lookup (test_init fixture skips its patch)",
78
81
  "slow: long-running; nightly/manual only",
79
82
  "llm: calls a paid LLM API; manual only, never CI",
80
83
  "slurm: needs a Slurm cluster; manual only, never CI",
@@ -7,7 +7,7 @@ runs the local loop elsewhere, reading placement from `~/.config/outerloop/.env`
7
7
  | Script | Purpose |
8
8
  | --- | --- |
9
9
  | `setup_branch_protection.sh` | Apply/refresh branch protection on `main`. Solo-phase default: 0 approvals, checks required, admins enforced. Re-run with `1` once a second code owner joins. |
10
- | `tick_chain.sbatch` | The Slurm tick entry: per-cadence chain (deploy, one tick, schedule the next) or, with `AUTORESEARCH_RESIDENT=1`, the resident loop (`docs/design/resident-tick.md`). |
10
+ | `tick_chain.sbatch` | The Slurm tick entry: per-cadence chain (deploy, one tick, schedule the next) or, with `OUTERLOOP_RESIDENT=1`, the resident loop (`docs/design/resident-tick.md`). |
11
11
  | `tick_deploy.sh` | The deploy step both modes source: stale-lock sweep, pull `main`, sync deps, read the operator's `.env` knobs, install backends. |
12
12
  | `tick_resident.sh` | The resident loop: deploy → one tick under a timeout → sleep to the slot, one `afterany:self` successor, pause exits clean, shim changes resubmit the successor. |
13
13
  | `requeue_moved_successors.sh` | Cancel pending same-name jobs that are no longer on the requested partition. The tick chain runs this before adding successors. |
@@ -9,7 +9,7 @@
9
9
  #
10
10
  # Usage: install_codex.sh [target_path]
11
11
  # target_path where to place the binary
12
- # (default: $AUTORESEARCH_CODEX_BIN, else ~/.local/bin/codex)
12
+ # (default: $OUTERLOOP_CODEX_BIN, else ~/.local/bin/codex)
13
13
  set -euo pipefail
14
14
 
15
15
  # Pinned: 0.130.0 is harness-verified — it needs neither the code-mode-host helper
@@ -20,7 +20,7 @@ WANT="0.130.0"
20
20
  # release asset can never execute on the host (integrity, not self-reported
21
21
  # version). To bump WANT: fetch the new asset and `sha256sum` it, then update both.
22
22
  WANT_SHA256="16779e7b7857508a768a36d7d4e084eec336ec23946ed70a9b09489b8f861190"
23
- TARGET="${1:-${AUTORESEARCH_CODEX_BIN:-$HOME/.local/bin/codex}}"
23
+ TARGET="${1:-${OUTERLOOP_CODEX_BIN:-$HOME/.local/bin/codex}}"
24
24
 
25
25
  have="$("$TARGET" --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1 || true)"
26
26
  if [ "$have" = "$WANT" ]; then