outerloop-science 0.2.0rc4__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/CHANGELOG.md +330 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/CITATION.cff +2 -2
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/CLAUDE.md +9 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/PKG-INFO +1 -1
- outerloop_science-0.3.0/RELEASING.md +69 -0
- outerloop_science-0.3.0/docs/design/accelerators.md +205 -0
- outerloop_science-0.3.0/docs/design/agent-environment.md +95 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/agent-protocols.md +1 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/architecture.md +19 -4
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/lifecycle.md +48 -11
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/meta.md +5 -2
- outerloop_science-0.3.0/docs/design/multi-cluster.md +464 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/research-loop-buildout.md +3 -2
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/roles.md +8 -6
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/scaling.md +5 -4
- outerloop_science-0.3.0/docs/endpoints.md +252 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/install.md +393 -12
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/roadmap.md +5 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/pyproject.toml +1 -0
- outerloop_science-0.3.0/scripts/install_bridge.sh +33 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/install_claude.sh +26 -12
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/install_codex.sh +40 -11
- outerloop_science-0.3.0/scripts/install_hermes.sh +109 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/tick_chain.sbatch +15 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/tick_deploy.sh +81 -73
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/tick_resident.sh +63 -27
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/__init__.py +1 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/attempt.py +1215 -386
- outerloop_science-0.3.0/src/outerloop/author_overrides.py +147 -0
- outerloop_science-0.3.0/src/outerloop/bridge_install.py +35 -0
- outerloop_science-0.3.0/src/outerloop/bridge_runtime/pyproject.toml +5 -0
- outerloop_science-0.3.0/src/outerloop/bridge_runtime/uv.lock +1811 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/brief.py +27 -11
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/cli.py +301 -23
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/climbboard.py +40 -15
- outerloop_science-0.3.0/src/outerloop/codex_bridge.py +529 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/compute.py +200 -9
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/contract.py +4 -6
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/dispatch.py +36 -2
- outerloop_science-0.3.0/src/outerloop/endpoint_wait.py +132 -0
- outerloop_science-0.3.0/src/outerloop/endpoints.py +261 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/github.py +96 -1
- outerloop_science-0.3.0/src/outerloop/gpu_lanes.py +92 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/harness.py +246 -43
- outerloop_science-0.3.0/src/outerloop/harness_cli.py +401 -0
- outerloop_science-0.3.0/src/outerloop/harness_pins.py +75 -0
- outerloop_science-0.3.0/src/outerloop/harnesses.toml +17 -0
- outerloop_science-0.3.0/src/outerloop/hermes_install.py +34 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/inbox.py +135 -12
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/init.py +335 -37
- outerloop_science-0.3.0/src/outerloop/instance.py +30 -0
- outerloop_science-0.3.0/src/outerloop/job_names.py +23 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/launchlog.py +14 -1
- outerloop_science-0.3.0/src/outerloop/ledger_branch.py +149 -0
- outerloop_science-0.3.0/src/outerloop/ledger_events.py +223 -0
- outerloop_science-0.3.0/src/outerloop/ledger_migrate.py +119 -0
- outerloop_science-0.3.0/src/outerloop/limits.py +151 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/measure.py +69 -13
- outerloop_science-0.3.0/src/outerloop/operator_limits.py +220 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/orchestrator.py +406 -87
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/panel.py +48 -4
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/paths.py +15 -0
- outerloop_science-0.3.0/src/outerloop/progress.py +317 -0
- outerloop_science-0.3.0/src/outerloop/provenance.py +33 -0
- outerloop_science-0.3.0/src/outerloop/rebind.py +229 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/review.py +7 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/review_agent.py +23 -6
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/review_agent_cli.py +34 -10
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/role_runner.py +42 -7
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/runstate.py +87 -8
- outerloop_science-0.3.0/src/outerloop/status.py +115 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/steward.py +60 -66
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/syscall.py +16 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/syscall_cli.py +16 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/tick.py +627 -124
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/verifier.py +14 -6
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/verify_agent_cli.py +12 -6
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/watcher.py +9 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/conftest.py +24 -0
- outerloop_science-0.3.0/tests/fixtures/README.md +80 -0
- outerloop_science-0.3.0/tests/fixtures/author_route_legacy.json +31 -0
- outerloop_science-0.3.0/tests/fixtures/author_route_missing_model.json +30 -0
- outerloop_science-0.3.0/tests/fixtures/author_route_missing_route.json +28 -0
- outerloop_science-0.3.0/tests/fixtures/base_moved_cc4e5d7.json +33 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/exit-code +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/provenance.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/stdout +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/completed/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/identity.json +4 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/provenance.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/dispatched_pre_gpu_identity/inflight/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
- outerloop_science-0.3.0/tests/fixtures/hermes_resume_legacy.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/hermes_sample_20260924.json +28 -0
- outerloop_science-0.3.0/tests/fixtures/launches-before-provenance.jsonl +2 -0
- outerloop_science-0.3.0/tests/fixtures/panel_wake_9ca3d7c.json +25 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/baselines/main@aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/brief.txt +44 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/end-request.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/ended.json +44 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-provenance/command.txt +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-provenance/job.sh +21 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/command.txt +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/exit-code +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/job.sh +22 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/stdout +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/eval-run/eval-candidate-aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa-f69684777d523fe6502ee4ed4d09f98410a56aec/submitted +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/generate.py.txt +72 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/identity.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/launches.jsonl +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/line.bundle +0 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/merged.json +44 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/open.json +44 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/pending/owner__repo.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/pending/owner__repo@agent-01.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/pr-states.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/queue.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/rebind.json +1 -0
- outerloop_science-0.3.0/tests/fixtures/rc1_v021/sessionless.json +44 -0
- outerloop_science-0.3.0/tests/ledger_fake.py +90 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_app_permissions.py +1 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_attempt.py +2106 -113
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_attempt_review.py +703 -85
- outerloop_science-0.3.0/tests/test_author_overrides.py +777 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_brief.py +1 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_climbboard.py +72 -3
- outerloop_science-0.3.0/tests/test_codex_bridge.py +831 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_compute.py +119 -0
- outerloop_science-0.3.0/tests/test_default_claude_model.py +588 -0
- outerloop_science-0.3.0/tests/test_endpoint_wait.py +256 -0
- outerloop_science-0.3.0/tests/test_endpoints.py +849 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_github.py +362 -0
- outerloop_science-0.3.0/tests/test_gpu_lanes.py +243 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_harness.py +39 -1
- outerloop_science-0.3.0/tests/test_harness_pins.py +600 -0
- outerloop_science-0.3.0/tests/test_hermes_author.py +195 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_hermes_harness.py +164 -12
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_inbox.py +357 -8
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_init.py +348 -1
- outerloop_science-0.3.0/tests/test_install_harness.py +557 -0
- outerloop_science-0.3.0/tests/test_instance.py +96 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_launchlog.py +25 -0
- outerloop_science-0.3.0/tests/test_ledger_events.py +650 -0
- outerloop_science-0.3.0/tests/test_ledger_migrate.py +139 -0
- outerloop_science-0.3.0/tests/test_limits.py +156 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_local_compute.py +26 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_measure.py +107 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_measure_and_decide.py +93 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_messages.py +27 -1
- outerloop_science-0.3.0/tests/test_operator_end.py +350 -0
- outerloop_science-0.3.0/tests/test_operator_limits.py +653 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_orchestrator.py +1165 -33
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_panel.py +39 -0
- outerloop_science-0.3.0/tests/test_paths.py +76 -0
- outerloop_science-0.3.0/tests/test_progress.py +186 -0
- outerloop_science-0.3.0/tests/test_rebind.py +774 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review.py +2 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review_agent.py +70 -1
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_runstate.py +13 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_start.py +155 -7
- outerloop_science-0.3.0/tests/test_status.py +123 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_steward.py +33 -32
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_syscall.py +34 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_syscall_cli.py +39 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_tick.py +268 -23
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_tick_resident.py +288 -5
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_verifier.py +7 -3
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_version.py +5 -0
- outerloop_science-0.2.0rc4/RELEASING.md +0 -38
- outerloop_science-0.2.0rc4/scripts/install_hermes.sh +0 -43
- outerloop_science-0.2.0rc4/src/outerloop/limits.py +0 -80
- outerloop_science-0.2.0rc4/src/outerloop/progress.py +0 -170
- outerloop_science-0.2.0rc4/tests/test_install_harness.py +0 -218
- outerloop_science-0.2.0rc4/tests/test_limits.py +0 -71
- outerloop_science-0.2.0rc4/tests/test_paths.py +0 -23
- outerloop_science-0.2.0rc4/tests/test_progress.py +0 -55
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/.gitignore +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/.pre-commit-config.yaml +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/.python-version +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/CONTRIBUTING.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/LICENSE +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/NOTICE +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/README.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/SECURITY.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/containers/README.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/containers/agent-py312.def +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/assets/icon-dark.svg +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/assets/icon-light.svg +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/assets/icon.svg +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/community.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/compute.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/contract.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/agent-substrate.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/base-reintegration.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/consolidation.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/dispatcher.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/eval-cache.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/external.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/github-app-auth.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/headline.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/judge-placement.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/onboarding.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/orchestrator-verify.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/public-surface.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/research-lines.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/research-loop.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/resident-tick.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/review-placement.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/reviewer-infra.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/role-cli.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/design/session-watcher.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/reviewer.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/docs/validation/author-syscalls.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/examples/review.yml +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/README.md +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/requeue_moved_successors.sh +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/setup_branch_protection.sh +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/scripts/sweep_git_locks.sh +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/__main__.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/appauth.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/appmanifest.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/contract_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/disk.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/evalcache.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/housekeeping.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/hypothesis.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/image.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/intake.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/maintain.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/maintain_agent_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/maintain_post_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/markers.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/posting.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/py.typed +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/review_post_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/review_summarize_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/roles.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/rolespec.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/style.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/verify_agent.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/src/outerloop/verify_post_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/fakes.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/helpers.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_appauth.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_appmanifest.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_bot_aliases.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_bot_login.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_channel_dir.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_codex_harness.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_contract.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_contract_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_contract_names.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_disk.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_dispatch.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_evalcache.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_hardening.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_housekeeping.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_image.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_import.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_intake.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_lifecycle_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_maintain.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_markers.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_packaging.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_posting.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_requeue_moved_successors.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review_agent_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review_hardening.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review_policy.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_review_summarize.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_role_runner.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_rolespec.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_sweep_git_locks.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_tick_chain_successors.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_tiers.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_verify_agent.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_verify_agent_cli.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/tests/test_watcher.py +0 -0
- {outerloop_science-0.2.0rc4 → outerloop_science-0.3.0}/uv.lock +0 -0
|
@@ -6,6 +6,333 @@ Versions follow [SemVer](https://semver.org).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.3.0] - 2026-10-05
|
|
10
|
+
|
|
11
|
+
### Upgrading
|
|
12
|
+
|
|
13
|
+
Operator actions (everything else needs no action; details in each entry):
|
|
14
|
+
|
|
15
|
+
- Claude Code is pinned at 2.1.285: run `outerloop harness upgrade claude`
|
|
16
|
+
(deploys that run harness upgrades do this on their own).
|
|
17
|
+
- Before selecting a chat-only Codex endpoint profile, install the bridge:
|
|
18
|
+
`outerloop harness upgrade --used`.
|
|
19
|
+
- Existing Hermes source-only installs: run `bash scripts/install_hermes.sh
|
|
20
|
+
"$REVIEW_HERMES_REPO"` (or full `outerloop init`) before Hermes sessions launch.
|
|
21
|
+
- Harness version overrides now need matching SHA-256 settings.
|
|
22
|
+
- Evals and baselines recorded by the previous kernel are measured again once
|
|
23
|
+
under the new cache key, with no extra charge.
|
|
24
|
+
- Parked authors keep the instructions they started with; judges use the new
|
|
25
|
+
rubric at once.
|
|
26
|
+
- Runs parked by the previous kernel resume after the upgrade.
|
|
27
|
+
- Before rolling back: consume pending rebind requests, finish capacity-parked
|
|
28
|
+
runs, runs with extended session limits, overridden and endpoint-routed runs,
|
|
29
|
+
and chat-only Codex sessions, and stop additional instances.
|
|
30
|
+
|
|
31
|
+
- A PR tests one idea. It may include the few changes that idea needs, and the
|
|
32
|
+
report states the effect of each change. Authors test several values in one
|
|
33
|
+
array launch and report the results around the chosen value; the panel may
|
|
34
|
+
block a single-value tuning change without them (new `landscape` finding
|
|
35
|
+
category). Upgrading: no action. Judges use the new rubric at once. An author
|
|
36
|
+
parked across the upgrade keeps the instructions it started with. Older readers
|
|
37
|
+
treat `landscape` as `other`. (This was listed under 0.2.1 by mistake; it
|
|
38
|
+
shipped after that tag.)
|
|
39
|
+
|
|
40
|
+
- Include GPU count and resolved GPU type in dispatched eval and baseline cache identity, including budget discounts. Legacy eval slots and baseline entries are cache misses. Upgrading: no action. An eval or baseline recorded by the previous kernel, in flight or finished, is measured again once under the new cache key, with no extra budget charge; pausing or draining before the upgrade does not avoid it.
|
|
41
|
+
|
|
42
|
+
- Park launch capacity refusals in capacity wait when an immediate resume is
|
|
43
|
+
unavailable or already refused, instead of ending the run. The next wake
|
|
44
|
+
delivers the refusal and a retry note, resuming the same session when supported
|
|
45
|
+
or starting a fresh session with the run context otherwise. Meters are
|
|
46
|
+
preserved and capacity waits do not exhaust stuck retries.
|
|
47
|
+
Upgrading: no action; existing records need no migration. A run parked by the
|
|
48
|
+
previous kernel in author-sleep with no session and no pending job starts a
|
|
49
|
+
fresh author leg on its next wake instead of ending as a session error.
|
|
50
|
+
Finish capacity-parked runs before rolling back; older kernels cannot resume
|
|
51
|
+
them.
|
|
52
|
+
|
|
53
|
+
- Add operator `outerloop rebind <run-id> [--root <root>] [--note <text>]` to
|
|
54
|
+
explicitly move an existing run to its slot's current author at the next leg,
|
|
55
|
+
including runs waiting on retired endpoints. Reports and board details show
|
|
56
|
+
ordered authors; evaluation and launch provenance retain the producing author.
|
|
57
|
+
Pending requests can be cancelled; three failed applications stop retries,
|
|
58
|
+
with the count and last error visible in status. A new request replaces a failed one.
|
|
59
|
+
- Upgrading: optional run fields `author_history`, `author_rebind_id` and
|
|
60
|
+
`stage.candidate_author` / `stage.candidate_authors`, plus `rebind.json` requests
|
|
61
|
+
and evaluation provenance, require no migration. Legacy author history is populated lazily. Consume
|
|
62
|
+
pending requests before rolling back to a kernel without rebind support.
|
|
63
|
+
|
|
64
|
+
- Add `outerloop end <run-id> [--root <root>] [--note <text>]` to request an
|
|
65
|
+
operator ending at the next tick, including runs waiting for an unavailable
|
|
66
|
+
endpoint. The tick uses the existing ending cleanup and refuses later publish.
|
|
67
|
+
- Upgrading: existing run records need no backfill; an absent `end-request.json`
|
|
68
|
+
means no request. Older kernels ignore requests and do not recognize the new
|
|
69
|
+
`operator` ending when writing records; keep the updated kernel for these runs.
|
|
70
|
+
|
|
71
|
+
- Author overrides accept operator-only `session_minutes` (10–240) and
|
|
72
|
+
`session_max_turns` (10–300). Limits bind with the author selection and survive
|
|
73
|
+
settings changes across wakes and review replies. Contracts can still lower
|
|
74
|
+
budgets; job walltime follows session duration and the operator job cap.
|
|
75
|
+
Judge budgets are unchanged.
|
|
76
|
+
- Upgrading: the new override fields are optional; existing settings and legacy
|
|
77
|
+
run records need no migration. Finish runs using extended limits before
|
|
78
|
+
rolling back to a version without bound author limits.
|
|
79
|
+
|
|
80
|
+
- Bad author overrides hold only affected fresh claims during ticks, with one log
|
|
81
|
+
per entry per tick; malformed settings hold named targets, or all fresh claims
|
|
82
|
+
when unreadable. Other tick services and bound runs continue; `outerloop start`
|
|
83
|
+
and `outerloop init` remain strict. No persisted state changes.
|
|
84
|
+
|
|
85
|
+
- Add read-only `outerloop status` (text/`--json`) for local runs and endpoint
|
|
86
|
+
outages. Endpoint waits stay out of the published board/status strip and never
|
|
87
|
+
trigger research-log commits; log one shared outage start and recovery with
|
|
88
|
+
duration and run IDs.
|
|
89
|
+
Validate optional served model and expiry in bounded endpoint address records.
|
|
90
|
+
- Upgrading: no action needed; the first endpoint deferral adds
|
|
91
|
+
`stage.endpoint_wait` and an `endpoint-waits/<profile>.json` log latch. Missing
|
|
92
|
+
keys/journals are tolerated; ended runs and in-flight PRs are unchanged.
|
|
93
|
+
Rollback to the preceding kernel safely ignores the additive state.
|
|
94
|
+
|
|
95
|
+
- Startup validation of `OUTERLOOP_AUTHOR_OVERRIDES` (`outerloop start`) uses the
|
|
96
|
+
image sessions actually run with, the default image when `OUTERLOOP_IMAGE` is unset. Before, a
|
|
97
|
+
codex override on a deployment without `OUTERLOOP_IMAGE` failed validation and stopped the tick.
|
|
98
|
+
|
|
99
|
+
- `OUTERLOOP_AUTHOR_OVERRIDES` accepts a list of entries per target, so different
|
|
100
|
+
slots of one target can use different authors (each listed entry names its slots;
|
|
101
|
+
a slot may appear only once). The single-object form is unchanged.
|
|
102
|
+
|
|
103
|
+
- Reject Codex authors and judges on chat-only endpoints during preflight when
|
|
104
|
+
the bridge runtime is missing or stale, with `outerloop harness upgrade --used`
|
|
105
|
+
as the fix, before spending the author budget.
|
|
106
|
+
|
|
107
|
+
- Codex authors and judges can use chat-only endpoint profiles through a
|
|
108
|
+
per-session LiteLLM bridge and streaming shim, including tool calls and
|
|
109
|
+
reasoning replay on resume. The bridge runtime is isolated, transitively
|
|
110
|
+
locked, installed by `outerloop harness upgrade`, and mounted read-only.
|
|
111
|
+
Direct Responses profiles keep their existing route and interruption behavior;
|
|
112
|
+
the Codex pin is unchanged. Bridge runtime defaults use `OUTERLOOP_CACHE_ROOT`,
|
|
113
|
+
else `$OUTERLOOP_ROOT/cache` (local fallback: `~/.outerloop/cache`). Harness
|
|
114
|
+
upgrades skip bridge detection for misconfigured roles with a diagnostic and
|
|
115
|
+
continue upgrading the other configured harnesses.
|
|
116
|
+
- Upgrading: install the bridge with `outerloop harness upgrade --used` before
|
|
117
|
+
selecting chat-only Codex profiles. Existing run records, session homes, and
|
|
118
|
+
harness retry records retain their formats; no backfill is needed. In-flight
|
|
119
|
+
direct-endpoint runs are unchanged. Before rollback, finish chat-only Codex
|
|
120
|
+
sessions or select a Responses-capable profile; older kernels reject that route.
|
|
121
|
+
|
|
122
|
+
- Research-line salvage and terminal snapshots now recheck scope admission
|
|
123
|
+
against the trusted contract before sealing. Out-of-scope changes are
|
|
124
|
+
dropped from the seal (tracked paths retain their parent content), while
|
|
125
|
+
admitted work and line memory survive normal endings and crashes after
|
|
126
|
+
scope refusals. Filtering leaves working files and the real index untouched
|
|
127
|
+
and logs a bounded list of dropped paths. Upgrading: no action; existing line
|
|
128
|
+
branches are left as they are, and the next snapshot drops out-of-scope paths.
|
|
129
|
+
Rollback is safe.
|
|
130
|
+
|
|
131
|
+
- Support separate instances on one cluster account: process-only absolute
|
|
132
|
+
`OUTERLOOP_ENV_FILE`, with the existing ownership/write-permission checks,
|
|
133
|
+
and stable settings-path suffixes for resident and per-cadence scheduler jobs.
|
|
134
|
+
Init, launch, deploy, harness status, and successor recovery use the selected
|
|
135
|
+
settings and instance identity.
|
|
136
|
+
- Upgrading: no action needed for the default fleet; its settings path and job
|
|
137
|
+
names remain unchanged. Existing run records, leases, and heartbeats need no
|
|
138
|
+
migration. Stop additional instances before rolling back to a version without
|
|
139
|
+
instance isolation.
|
|
140
|
+
|
|
141
|
+
- Launch, submit, and stale-submit checkpoint scope violations now refuse every
|
|
142
|
+
request and resume the author with the offending paths and bounded allowed
|
|
143
|
+
scope in the kernel inbox. Later refusals say “Refused again:”. Refusals
|
|
144
|
+
repeat and never end the run; session walltime and contract sleep, launch,
|
|
145
|
+
and GPU-hour budgets bound the loop. Refusal seals nothing, runs no jobs or
|
|
146
|
+
measurements, and spends no request budget. The authoritative measurement
|
|
147
|
+
scope check remains terminal. Abandoning a refused tree ends normally
|
|
148
|
+
without measurement, a line snapshot, or a push; outage and budget endings
|
|
149
|
+
also preserve the rejection. Scope admission precedes malformed request
|
|
150
|
+
and budget refusals.
|
|
151
|
+
- Upgrading: no action or backfill needed. Scope refusals use the existing
|
|
152
|
+
kernel note payload and refusal keys; old inboxes, ended runs, and in-flight
|
|
153
|
+
runs remain readable without rewriting delivered messages. The next request
|
|
154
|
+
uses the new admission behavior. The rejection flag is in-memory only;
|
|
155
|
+
no run-record fields change; rollback is
|
|
156
|
+
safe and restores terminal scope admission checks.
|
|
157
|
+
|
|
158
|
+
- Deployment author overrides select a backend/model per target or agent slot,
|
|
159
|
+
bind it to each run, and leave panel and CI reviewer inheritance on the fleet
|
|
160
|
+
author. Per-run board details identify the effective author and overrides.
|
|
161
|
+
- Endpoint profiles accept an absolute `URL_FILE` instead of `URL`, reading bare
|
|
162
|
+
URLs or JSON addresses on every session. Missing files, failed bounded health
|
|
163
|
+
checks, and interrupted checks defer sessions without spending wake retries.
|
|
164
|
+
Init provisions fleet and override backends before validating prerequisites.
|
|
165
|
+
- Upgrading: no action or backfill needed; records without `author_overridden`
|
|
166
|
+
retain their existing author and panel behavior, including ended runs. New
|
|
167
|
+
overridden records reuse the saved author route and add this optional flag.
|
|
168
|
+
Rollback reads the records but loses fleet-only panel inheritance for overrides;
|
|
169
|
+
finish overridden runs and fresh endpoint capacity parks before rolling back.
|
|
170
|
+
Fresh endpoint deferrals reuse author-sleep capacity parks without a session ID;
|
|
171
|
+
older kernels cannot resume those parks.
|
|
172
|
+
|
|
173
|
+
- Launch ledger submissions require dispatched launch job IDs, preventing stale
|
|
174
|
+
checkpoints from attributing discarded launches to a commit. Gate capacity
|
|
175
|
+
waits still record any dispatched sibling launches. Existing ledger rows and
|
|
176
|
+
run records remain readable and unchanged; no schema change or backfill.
|
|
177
|
+
|
|
178
|
+
- PR measurement tables identify the base and candidate commits and the shared
|
|
179
|
+
eval command from the base tree; experiment tables identify launch commits.
|
|
180
|
+
Verify and review briefs caution against comparing numbers across commits
|
|
181
|
+
without checking history.
|
|
182
|
+
- Upgrading: no action or backfill needed; the first tick tolerates launch ledger
|
|
183
|
+
rows without `commit` (shown as unknown), including ended runs and in-flight
|
|
184
|
+
PRs. New launch records include the sealed commit; existing records and PR
|
|
185
|
+
bodies are not rewritten. Rollback is safe: older readers ignore the added
|
|
186
|
+
field.
|
|
187
|
+
|
|
188
|
+
- Hermes author resumes that exceed the replay budget stay parked with a
|
|
189
|
+
configuration-blocked status, retaining their session and snapshot without
|
|
190
|
+
consuming wake retries. Author/judge separation checks effective key paths
|
|
191
|
+
and credential values before constructing sessions; init rejects incomplete
|
|
192
|
+
Hermes configuration only for Hermes authors. Endpoint author credentials are
|
|
193
|
+
shared by attempt and tick preflight, including native panel key comparisons.
|
|
194
|
+
- Upgrading: no action or backfill needed; legacy records without
|
|
195
|
+
`stage.hermes_resume_required_chars` are unblocked. The first oversized wake
|
|
196
|
+
records the required budget; raising `OUTERLOOP_HERMES_RESUME_MAX_CHARS` lets
|
|
197
|
+
the next tick or wake resume. Successful resumes and endings clear the marker,
|
|
198
|
+
so later normal sleeps are not configuration wakes. Before rollback, resolve
|
|
199
|
+
blocked runs: older kernels ignore this optional field and may consume retries
|
|
200
|
+
or abort them.
|
|
201
|
+
|
|
202
|
+
- Hermes is an author peer: init, native provider/endpoint validation, contained
|
|
203
|
+
fresh and resumed sessions, absolute syscall commands, and separate author keys.
|
|
204
|
+
Resume replay preserves the original brief and latest results within
|
|
205
|
+
`OUTERLOOP_HERMES_RESUME_MAX_CHARS` (default 120000), with explicit omission counts.
|
|
206
|
+
- Upgrading: no backfill; existing records and full saved transcripts remain
|
|
207
|
+
readable. The first Hermes wake applies the replay bound. Finish Hermes author
|
|
208
|
+
runs before rollback: older kernels reject unsupported Hermes author wakes
|
|
209
|
+
(the endpoint-profile predecessor accepts endpoint routes only). Ended records
|
|
210
|
+
remain readable. See [Hermes setup and compatibility](docs/install.md).
|
|
211
|
+
|
|
212
|
+
- Endpoint profiles declare compatible APIs and use `model[endpoint=profile]`
|
|
213
|
+
selectors, preserving native vendor model IDs. Judge credentials enforce file
|
|
214
|
+
separation; verdicts redact the session key before posting or aggregation.
|
|
215
|
+
- Upgrading: legacy records missing model/route fields retain native routing;
|
|
216
|
+
missing native model configuration fails closed instead of adopting fleet endpoints.
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
- Named, file-authenticated endpoint profiles work for authors, panel lenses,
|
|
220
|
+
and standalone reviewers on Claude Code, Codex, and Hermes. Sessions use each
|
|
221
|
+
backend's native API configuration; keys reach contained sessions through env,
|
|
222
|
+
never argv. See [endpoint settings and validation](docs/endpoints.md).
|
|
223
|
+
- Hermes is pinned to v2026.9.24 (`f97608f178d1ffeca59860195ab7da295f7c8e5f`).
|
|
224
|
+
Remove Fire quoting for its new argparse entrypoint, enable
|
|
225
|
+
`model.reasoning_echo` for endpoint profiles, and sanitize its instruction-file
|
|
226
|
+
aliases in judge checkouts.
|
|
227
|
+
- Upgrading: existing runs need no backfill; the first tick validates endpoint
|
|
228
|
+
selections without changing legacy routes. New endpoint authors save
|
|
229
|
+
`model[endpoint=profile]` selectors; keep those profiles until runs finish, and finish
|
|
230
|
+
endpoint runs before rolling back. Hermes source/runtime must be upgraded
|
|
231
|
+
with the harness. See [compatibility and rollback](docs/endpoints.md#upgrade-and-rollback).
|
|
232
|
+
|
|
233
|
+
### Fixed
|
|
234
|
+
|
|
235
|
+
- Manual harness upgrades honor `--root`, environment, and `.env` state roots. Retry records are replaced atomically; unreadable or invalid records are logged and ignored. Deploy loads the configured cache root before selecting the uv cache.
|
|
236
|
+
- Harness installers reinstall changed binaries rather than refusing repair; Codex checks its installed binary digest separately from the archive pin, and Hermes runtime reuse checks the interpreter digest.
|
|
237
|
+
|
|
238
|
+
- Harness upgrades enforce a per-harness deadline, kill timed-out installer process groups, restrict installer environments, and back off failed pins while deleting failed candidates. Overrides require explicit integrity hashes; installed binaries are hash-checked before reuse. Workflow pin resolution fails explicitly on older reviewer refs without a pins reader.
|
|
239
|
+
|
|
240
|
+
- Tick entrypoints export a state-root uv cache before running Python. Fleet job environments preserve explicit cache paths and default per-user caches below `OUTERLOOP_CACHE_ROOT` (or the state root).
|
|
241
|
+
|
|
242
|
+
- Run-owned launch, evaluation, and wake job names stay within 128 characters, retaining a stable run key when shortened. Normal names stay unchanged; GPU usage, queue attribution, evaluation deduplication, and flight retention recognize the bounded names.
|
|
243
|
+
- Intake admission counts queued attempts using the existing pending markers, including jobs queued beyond the marker TTL.
|
|
244
|
+
|
|
245
|
+
- GPU accounting accepts Slurm 25.05 wrapped numeric fields and legacy integers, recognizes typed GPU requests and per-node counts, and limits pending array remainders to available throttle slots.
|
|
246
|
+
- CI Hermes provisioning uses the shared runtime installer with anonymous clone retries and matching workflow pins. A source-specific lock protects checkout and runtime mutations; Python discovery excludes active virtualenvs.
|
|
247
|
+
- Panel preflight checks Hermes runtime readiness and explains installation; full init preserves review model and provider settings with environment precedence.
|
|
248
|
+
|
|
249
|
+
- Contained Hermes sessions can start with read-only source; sessions no longer reinstall dependencies or attempt an editable project build.
|
|
250
|
+
|
|
251
|
+
### Added
|
|
252
|
+
|
|
253
|
+
- Optional per-target GPU lanes route evals and author launches to deployment-specific partitions, accounts, GPU types, and sbatch flags.
|
|
254
|
+
|
|
255
|
+
- Packaged `harnesses.toml` owns Claude, Codex, and Hermes pins. `outerloop harness status` reports installed versions, paths, drift, and operator overrides; `harness upgrade [name...]` verifies versioned installations before atomically recording their paths. Successful kernel deploys upgrade only configured backends; failures retain the previous installation.
|
|
256
|
+
|
|
257
|
+
- Live, tighten-only `<root>/limits.toml` GPU and active-attempt ceilings, with global defaults and per-target sections. Scheduler-reported GPU usage covers pending and running experiments, sweeps, evaluations, and GPU-bearing sessions. Authors receive uncharged launch refusals; evaluations wait for capacity. Lowering a ceiling does not cancel jobs.
|
|
258
|
+
- Read-only `outerloop limits` reports operator ceilings and fleet-owned running/pending GPU usage.
|
|
259
|
+
|
|
260
|
+
### Changed
|
|
261
|
+
|
|
262
|
+
- Claude Code pin 2.1.272 -> 2.1.285, the current stable release;
|
|
263
|
+
`claude-opus-5-5` refuses Claude Code older than 2.1.280.
|
|
264
|
+
- Upgrading: legacy Codex archive-only markers and Hermes runtimes without interpreter digests are reinstalled on upgrade; legacy Hermes runtimes remain launchable. Existing retry records remain readable, and corrupt records are treated as empty. Run state and in-flight PRs are unchanged; rollback leaves the additional digest files unused.
|
|
265
|
+
|
|
266
|
+
- Upgrading: version overrides now require matching SHA-256 settings; legacy Codex installs without hash markers are reprovisioned. New retry state and hash markers are ignored by older kernels; the first successfully synced tick verifies configured harnesses and records new paths only when needed. Legacy `.env` paths and Hermes runtimes remain readable; old artifacts are retained. See `docs/install.md` for rollback across kernel pins.
|
|
267
|
+
|
|
268
|
+
- Hermes installs a standalone Python and venv once per pinned commit in a sibling runtime, then launches Python directly. Full `init` provisions configured Hermes judges and records their source path; `--no-install-harness` opts out.
|
|
269
|
+
|
|
270
|
+
### Upgrading notes for the entries above
|
|
271
|
+
|
|
272
|
+
- No action needed; OUTERLOOP_GPU_LANES is optional.
|
|
273
|
+
|
|
274
|
+
- Upgrading: full run-ID names and legacy 60-character queue names remain readable; shortened names use a derived run key without changing run records. Intake adds `@intake-<issue>` files in the existing pending directory; legacy unsuffixed and agent-slot markers remain readable. Drain queued intake jobs from older submitters (which wrote no marker) before relying on attempt ceilings. Upgrade all kernels together; older kernels do not recognize shortened names or intake markers, so drain those jobs before rollback.
|
|
275
|
+
|
|
276
|
+
- Upgrading: the optional `stage.capacity_wait` flag tolerates missing fields; existing state records need only their target for scheduler attribution. No contract schema change or admission journal. Drain older jobs whose names omit the full run ID (and older local jobs without scheduler metadata), and upgrade all submitters before relying on ceilings. Concurrent admissions may overshoot by one batch for two simultaneous checks; no cross-node admission lock.
|
|
277
|
+
- Upgrading: existing Hermes source-only installs require `bash scripts/install_hermes.sh "$REVIEW_HERMES_REPO"` (or full `outerloop init --force` with Hermes configured) to create the persisted runtime. Run records and resume transcripts are unchanged; rollback leaves the sibling runtime unused.
|
|
278
|
+
|
|
279
|
+
## [0.2.1] - 2026-09-25
|
|
280
|
+
|
|
281
|
+
### Upgrading
|
|
282
|
+
|
|
283
|
+
- No action needed; an author's report-only answer to the panel now updates the PR, and a PR whose panel clears is marked ready for review.
|
|
284
|
+
- No action needed; a PR held only by a base-moved blessing heals on the next tick.
|
|
285
|
+
- No contract change; existing contract files need no edits.
|
|
286
|
+
- The `research-log` ledger branch is created from the default branch on first publish if it is missing.
|
|
287
|
+
- An existing `research-log` branch may carry a stale copy of main's ledger, so check `BENCHMARKS.md` there. If it is missing, run `outerloop migrate-ledger --target OWNER/REPO --main-sha <current main sha> --dry-run`, then repeat without `--dry-run`. If it is stale, add `--force` to both runs; it replaces the whole table and erases confirmed rows, so look at the existing table first.
|
|
288
|
+
- Let open agent PRs finish before upgrading: PRs published by 0.2.0 carry a ledger commit and have no pending record, so their merge is not recorded on `research-log`.
|
|
289
|
+
- No action is needed for authors parked under the old base-moved wording; they receive the corrected fold message once when the kernel next detects that their branch lacks the current base tip.
|
|
290
|
+
- No action is needed for saved gate results without a base commit; they are ignored and the candidate must be measured again.
|
|
291
|
+
- Parked runs keep their recorded author backend and model; resumed panels now inherit from that author. Set an explicit model for any panel lens using another backend.
|
|
292
|
+
- Set `OUTERLOOP_CLAUDE_MODEL=<model>` in `.env` if any Claude role lacks an explicit or inherited model, including the steward when its key is configured.
|
|
293
|
+
- Restart local loops after installing; they do not auto-update or reload `.env`.
|
|
294
|
+
- No action is needed for inbox messages already delivered: kernel messages written before this version stay readable, and any output they quote stays fenced as data.
|
|
295
|
+
- No action is needed for `end` requests already staged: old requests stay valid, and a saved withdrawal is finished by the next tick.
|
|
296
|
+
|
|
297
|
+
### Fixed
|
|
298
|
+
|
|
299
|
+
- Report-only submissions update the PR report and panel verdict without pushing a commit.
|
|
300
|
+
- Draft PRs are marked ready for review when the latest panel read clears. Blocking draft banners tell the author to address the findings.
|
|
301
|
+
- Research line improvements can be blessed when their measured base contains the current base tip.
|
|
302
|
+
- Messages the kernel writes itself now render as its instructions; output the kernel quotes from elsewhere stays fenced as data.
|
|
303
|
+
- Authors are now told they may fold a moved base with one merge commit, including resolving conflicts within that merge.
|
|
304
|
+
- A submit made after the base moves is saved before any gate measurement or staged experiments run. The author is told to fold the base and submit again, or to submit again after the measurement base is refreshed.
|
|
305
|
+
- After folding the base, an author is told to submit directly and repeat its experiment only if the new base changes its hypothesis. PR updates and saved snapshots retain the measured base in their history, and a saved gate result is reused only for the same base and code.
|
|
306
|
+
- Updates brought in from the base no longer count as the author's changes when checking a resumed PR's allowed files. Authors receive a reason when an attempt ends but its open PR leaves the run parked for further work.
|
|
307
|
+
- Authors are woken and automatic merging is held when their PR branch does not contain the current base tip, including when conflicts block merging.
|
|
308
|
+
- Resident tick jobs replace vanished or finished successors and check the successor before handing over. If recovery fails, the current job keeps ticking through the reserved handover time until its time limit.
|
|
309
|
+
|
|
310
|
+
### Added
|
|
311
|
+
|
|
312
|
+
- Authors can withdraw a superseded open PR with `end --withdraw "<reason>"`; the kernel closes it with the reason and ends the run.
|
|
313
|
+
- `outerloop migrate-ledger --target OWNER/REPO --main-sha SHA [--dry-run] [--force]` imports the ledger from the specified current main commit into `research-log`.
|
|
314
|
+
- Full `outerloop init` requires and writes a Claude model even for a deployment without Claude roles, using `--claude-model`, then the shell's `OUTERLOOP_CLAUDE_MODEL`, then a required interactive prompt. A focused `init --github-app` run preserves existing `.env` settings it does not manage.
|
|
315
|
+
|
|
316
|
+
### Changed
|
|
317
|
+
|
|
318
|
+
- An idea with a clear mechanism that does not yet beat the best is reported as a success and kept on the author's research line.
|
|
319
|
+
- The sweep rechecks ancestry for PRs held only by a base-moved blessing.
|
|
320
|
+
- Merges performed by the sweep are observed and confirmed in the same tick.
|
|
321
|
+
- Scope checks compare the full candidate tree with its merge-base against the fetched base tip; changes that landed on the base branch never count as the author's, and author edits to the ledger files are refused like any other out-of-scope path.
|
|
322
|
+
- Publish is refused when the candidate shares no history with the base; fetch the base and fold it before submitting again.
|
|
323
|
+
- The leaderboard moves from PR branches and main to the `research-log` branch.
|
|
324
|
+
- Results are pending when published and confirmed when the kernel observes their PR merge, provided the final PR head and merged files match the measurement.
|
|
325
|
+
- The leaderboard shows the main commit at which each result was confirmed; imported rows say their provenance is unknown.
|
|
326
|
+
- PRs carry only the measured files; publication no longer adds a ledger commit.
|
|
327
|
+
- **Breaking:** The built-in Claude model default is gone; set `OUTERLOOP_CLAUDE_MODEL=<model>` in `.env` when a Claude role lacks an explicit or inherited model, including the steward when its key is configured. `outerloop start` refuses a missing Claude setting only in that case, so a deployment with no Claude roles can leave it unset.
|
|
328
|
+
- **Breaking:** Panel lenses without a backend now use the author's backend, and lenses without a model inherit the author's model on that backend. Lenses using another backend require an explicit model, even if `OUTERLOOP_CLAUDE_MODEL` is set.
|
|
329
|
+
- **Breaking:** The review and verify Actions use their model input or fall back to the `OUTERLOOP_CLAUDE_MODEL` Actions variable for Claude. With neither configured, the verifier logs a warning and skips verification with a successful exit.
|
|
330
|
+
- Start, tick checks and running panels now resolve panel models using the same rules. Resumed panels use the parked run's recorded author backend and model.
|
|
331
|
+
|
|
332
|
+
## [0.2.0] - 2026-09-18
|
|
333
|
+
|
|
334
|
+
**Upgrade note.** After upgrading run `outerloop permissions --open`: the App needs `checks: read` and `actions: read` for check results to reach authors as messages. This release redesigns the run lifecycle. It adds three run states, one inbox, `end`, review top-ups, and wakes that run by default. It also removes the old `AUTORESEARCH_*` names.
|
|
335
|
+
|
|
9
336
|
Authors now use `message` for public posts, reminders to self and messages to
|
|
10
337
|
live agents on the same target. Sibling messages keep a sent copy, and inbox
|
|
11
338
|
headers name both parties with local message numbers. `--reply-to` links a
|
|
@@ -22,6 +349,8 @@ tasks to finish.
|
|
|
22
349
|
|
|
23
350
|
### Added
|
|
24
351
|
|
|
352
|
+
- `OUTERLOOP_CLAUDE_MODEL` configures the shared default for all Claude roles, including deployments whose Vertex project has access to a different model. `OUTERLOOP_AUTHOR_MODEL` still overrides the author default. Vertex auxiliary fast calls default to the session model, avoiding dependencies on models the project has not enabled; `OUTERLOOP_VERTEX_SMALL_MODEL` overrides that default.
|
|
353
|
+
|
|
25
354
|
- `outerloop init` installs a missing author CLI and records its path; `--no-install-harness` opts out. Claude has a pinned, SHA256-verified installer.
|
|
26
355
|
|
|
27
356
|
- `OUTERLOOP_TICK_HOST=login` (or `--tick-host login`) runs a niced foreground tick loop against Slurm, with one tick lease per state root.
|
|
@@ -79,7 +408,7 @@ tasks to finish.
|
|
|
79
408
|
|
|
80
409
|
- Submit works in review and fast-forwards the PR after a confirmed auto-merge disarm; the measured number is posted first, even if publication is refused. Each review leg measures against its freshly fetched base. Verdicts, findings and publish refusals reach the author as messages. A reply staged during a review leg suppresses its final-text comment. A failed submitted park ends as negative-result only when no author session can resume to receive its verdict; its report and notebook are saved and its issue claim released. Sessions without the tool (no launcher or no resume support) are still measured at finish and may end on the verdict by design. Submit needs no prior launch or report. A session offered submit that stops without it ends unmeasured, or returns to review if it has a PR. Legacy follow-up re-measures retire on their next wake and release their snapshots.
|
|
81
410
|
|
|
82
|
-
- Authors can post replies through `
|
|
411
|
+
- Authors can post replies through `message` and launch experiments or sleep while a PR is in review, using the run’s remaining budget. Comments and base moves received while parked reach the author at its next wake. Replies are kept in an outbox for retries, review launches check committed edits, and closing a run holds its wake lease. Review edits are measured and published only on submit, using the same remaining GPU budget as other author work.
|
|
83
412
|
|
|
84
413
|
- Wake messages now use one inbox and one renderer. Launch results, gate verdicts, panel findings, review comments and base moves reach the session in the same fenced format, from files kept beside the run's record. Advisory panel findings now reach the author alongside blocking ones. The sibling view is refreshed at every wake of a parked author session. The kernel's wake text states facts; the research advice it used to carry is gone.
|
|
85
414
|
|
|
@@ -22,6 +22,11 @@ uv run pre-commit run --all-files
|
|
|
22
22
|
`.github/` are forbidden write paths everywhere, regardless of contract YAML.
|
|
23
23
|
- Budget caps are load-bearing safety features, not tunables to raise casually.
|
|
24
24
|
- Never commit credentials, transcripts, or run artifacts (SECURITY.md).
|
|
25
|
+
- This repository is public: code, tests, docs, commit messages and PR
|
|
26
|
+
descriptions carry no deployment specifics (cluster, account, partition or
|
|
27
|
+
node names, or what hardware a lab has). Use generic examples such as
|
|
28
|
+
`owner/repo`, `my-account`, `gpu-large`, and motivate a change by the general
|
|
29
|
+
need.
|
|
25
30
|
- Merge commits only; never rebase, squash, or force-push.
|
|
26
31
|
- **Review until quiet**: development PRs iterate advisory-review rounds
|
|
27
32
|
(after a fix commit, remove then re-add the `autoresearch:review` label
|
|
@@ -34,3 +39,7 @@ uv run pre-commit run --all-files
|
|
|
34
39
|
green is not read.
|
|
35
40
|
- Imports are absolute (`from autoresearch...`); deps go in with their code +
|
|
36
41
|
`uv lock`; CHANGELOG under `[Unreleased]`.
|
|
42
|
+
- A PR that changes state read across kernel versions (run records, PR
|
|
43
|
+
branches, ledger files, inbox messages, caches) carries a compatibility
|
|
44
|
+
statement, a legacy fixture, a backfill or tolerance, and an `Upgrading:`
|
|
45
|
+
changelog line (RELEASING.md).
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: outerloop-science
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
|
|
5
5
|
Project-URL: Homepage, https://outerloop.science
|
|
6
6
|
Project-URL: Repository, https://github.com/outerloop-science/outerloop
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
# Releasing
|
|
2
|
+
|
|
3
|
+
## Versioning
|
|
4
|
+
|
|
5
|
+
- SemVer 0.x; single source `src/outerloop/__init__.py`; tags `vX.Y.Z`;
|
|
6
|
+
Keep-a-Changelog. Pre-releases use PEP 440 suffixes (`0.1.0.dev0`,
|
|
7
|
+
`0.1.0rc1`) and are tagged the same way (`v0.1.0.dev0`).
|
|
8
|
+
|
|
9
|
+
## State that outlives a release
|
|
10
|
+
|
|
11
|
+
Some state is written by one kernel version and read by the next: run
|
|
12
|
+
records and their stage keys, PR branches and the publish journal, snapshot
|
|
13
|
+
and line refs, the research-log ledger and its retry intents, inbox messages
|
|
14
|
+
and their deduplication keys, syscall staging files, measurement caches, the
|
|
15
|
+
launch journal, PR-body and comment markers. A PR that changes how any of
|
|
16
|
+
these is written or read includes four things:
|
|
17
|
+
|
|
18
|
+
1. A compatibility statement in the PR body: which surfaces change, the
|
|
19
|
+
oldest state still read, what happens to runs and PRs in flight on the
|
|
20
|
+
first tick after the upgrade, and whether rolling back is safe.
|
|
21
|
+
2. A fixture for each surface the PR changes (a run record, an inbox, a
|
|
22
|
+
branch layout), produced by every release whose state the new code still
|
|
23
|
+
reads, exercised through the new code: first pass, a second idempotent
|
|
24
|
+
pass, and a retry after an interruption.
|
|
25
|
+
3. A backfill or an explicit tolerance for the old state, including missing
|
|
26
|
+
fields and ended runs. If a case is not supported, the PR says so instead
|
|
27
|
+
of leaving it to the operator to discover.
|
|
28
|
+
4. One `Upgrading:` line in the changelog: "no action needed; the first tick
|
|
29
|
+
does X", or the exact operator command. Automatic migration is claimed
|
|
30
|
+
only when the fixture proves it.
|
|
31
|
+
|
|
32
|
+
Text an agent has already received is state too. Changing a message's
|
|
33
|
+
wording does not reach a parked run unless its deduplication key changes.
|
|
34
|
+
|
|
35
|
+
## Cutting a release
|
|
36
|
+
|
|
37
|
+
1. List the merged PRs since the last tag that touch the state above and
|
|
38
|
+
check each has the four items. Collect their `Upgrading:` lines into one
|
|
39
|
+
section at the top of the release's changelog entry.
|
|
40
|
+
2. PR: bump `__version__`, set `CITATION.cff`'s `version` and `date-released`,
|
|
41
|
+
and move the `[Unreleased]` entries under the new version. Before tagging,
|
|
42
|
+
run the fixtures from step 1 once more against this PR's head, covering a
|
|
43
|
+
run with an open PR, a run whose PR merged, and an ended run. A dev or rc pre-release still bumps the version (PyPI never
|
|
44
|
+
accepts a version twice, so the next one is `.dev1`, `rc2`, ...) but leaves
|
|
45
|
+
`[Unreleased]` in place until the final release.
|
|
46
|
+
3. `git tag vX.Y.Z && git push origin vX.Y.Z`. The `release` workflow builds
|
|
47
|
+
and publishes `outerloop-science` to PyPI through Trusted Publishing; the
|
|
48
|
+
one-time PyPI setup is described at the top of
|
|
49
|
+
`.github/workflows/release.yml`.
|
|
50
|
+
4. Once the `release` workflow run for the tag is green (`gh run watch` on
|
|
51
|
+
it), `pip install outerloop-science==X.Y.Z` in a fresh venv, then
|
|
52
|
+
`outerloop --help`. This comes before the GitHub release: publishing it is
|
|
53
|
+
what Discord announces, so nothing is announced that did not install.
|
|
54
|
+
5. `gh release create vX.Y.Z --generate-notes`, with `--prerelease` for a dev
|
|
55
|
+
or rc tag.
|
|
56
|
+
6. Announce. Discord's `#announcements` gets the release from the GitHub
|
|
57
|
+
webhook on its own (docs/community.md). For a final release, also write
|
|
58
|
+
the post for X (`@outerloop_sci`) and Bluesky (`@outerloop.science`): one
|
|
59
|
+
or two sentences on what changed for the reader, the install line, the
|
|
60
|
+
link to the release. Dev and rc pre-releases are not posted to X or
|
|
61
|
+
Bluesky; Discord gets them through the webhook like any release.
|
|
62
|
+
|
|
63
|
+
## Public repo
|
|
64
|
+
|
|
65
|
+
Public since 2026-09-05. Secret scanning and push protection are on, and
|
|
66
|
+
`main` is protected by `scripts/setup_branch_protection.sh` (pull request
|
|
67
|
+
required, the `ci` check, conversations resolved, no force-push, admins
|
|
68
|
+
included). History is immutable now; prevention (gitleaks in pre-commit and
|
|
69
|
+
CI, push protection) is the real defense. `CITATION.cff` is still owed.
|