outerloop-science 0.1.0.dev3__tar.gz → 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/CHANGELOG.md +117 -0
- outerloop_science-0.1.1/CITATION.cff +17 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/PKG-INFO +7 -2
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/README.md +1 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/RELEASING.md +2 -2
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/architecture.md +5 -5
- outerloop_science-0.1.1/docs/design/base-reintegration.md +92 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/onboarding.md +5 -5
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/reviewer-infra.md +3 -3
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/role-cli.md +1 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/install.md +30 -5
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/validation/author-syscalls.md +1 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/pyproject.toml +7 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/__init__.py +1 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/appauth.py +17 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/attempt.py +100 -15
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/cli.py +64 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/climbboard.py +64 -22
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/followup.py +2 -9
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/github.py +32 -12
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/harness.py +21 -29
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/init.py +26 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/maintain.py +29 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/maintain_post_cli.py +4 -4
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/review_summarize_cli.py +1 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/steward.py +2 -9
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/tick.py +184 -135
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_appauth.py +19 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_attempt.py +130 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_followup.py +25 -10
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_github.py +17 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_init.py +46 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_maintain.py +17 -5
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_start.py +99 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_tick.py +160 -1
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/uv.lock +60 -60
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/.gitignore +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/.pre-commit-config.yaml +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/.python-version +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/CLAUDE.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/CONTRIBUTING.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/LICENSE +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/NOTICE +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/SECURITY.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/containers/README.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/containers/agent-py312.def +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/assets/icon-dark.svg +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/assets/icon-light.svg +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/assets/icon.svg +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/community.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/compute.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/contract.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/agent-substrate.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/consolidation.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/dispatcher.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/eval-cache.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/external.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/github-app-auth.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/headline.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/judge-placement.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/meta.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/orchestrator-verify.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/public-surface.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/research-lines.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/research-loop-buildout.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/research-loop.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/resident-tick.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/review-placement.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/roles.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/scaling.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/design/session-watcher.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/reviewer.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/docs/roadmap.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/examples/review.yml +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/README.md +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/install_codex.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/install_hermes.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/requeue_moved_successors.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/setup_branch_protection.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/sweep_git_locks.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/tick_chain.sbatch +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/tick_deploy.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/scripts/tick_resident.sh +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/__main__.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/appmanifest.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/brief.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/compute.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/contract.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/contract_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/disk.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/dispatch.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/evalcache.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/housekeeping.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/image.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/intake.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/launchlog.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/limits.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/maintain_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/markers.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/measure.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/orchestrator.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/panel.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/paths.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/posting.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/progress.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/py.typed +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/review.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/review_agent.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/review_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/review_post_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/role_runner.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/roles.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/rolespec.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/runstate.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/style.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/syscall.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/syscall_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/verifier.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/verify_agent.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/verify_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/verify_post_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/src/outerloop/watcher.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/conftest.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/fakes.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/helpers.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_appmanifest.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_bot_aliases.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_bot_login.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_brief.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_channel_dir.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_climbboard.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_codex_harness.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_compute.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_contract.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_contract_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_contract_names.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_disk.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_dispatch.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_env_bridge.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_evalcache.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_hardening.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_harness.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_hermes_harness.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_housekeeping.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_image.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_import.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_intake.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_launchlog.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_limits.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_local_compute.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_markers.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_measure.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_measure_and_decide.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_orchestrator.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_packaging.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_panel.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_paths.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_posting.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_progress.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_requeue_moved_successors.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review_agent.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review_hardening.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review_policy.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_review_summarize.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_role_runner.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_rolespec.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_runstate.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_steward.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_sweep_git_locks.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_syscall.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_syscall_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_tick_chain_successors.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_tick_resident.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_tiers.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_verifier.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_verify_agent.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_verify_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_version.py +0 -0
- {outerloop_science-0.1.0.dev3 → outerloop_science-0.1.1}/tests/test_watcher.py +0 -0
|
@@ -6,8 +6,112 @@ Versions follow [SemVer](https://semver.org).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.1] - 2026-09-11
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
|
|
13
|
+
- Wakes never recorded a launch as ended. A nested `import time` in the wake function shadowed the module import, so the ledger step failed on every wake, and the failure was only logged to a stderr the scheduler discards. The PR's experiments table stayed empty and `history` showed every launch as not back. The nested imports are gone, the best-effort wrapper now logs the redacted traceback, and a wake writes its kernel log to `runs/<run>/kernel.log`.
|
|
14
|
+
|
|
15
|
+
### Changed
|
|
16
|
+
|
|
17
|
+
- The PyPI project page and the Python-version badge now read correctly: the package metadata lists the supported Python version (3.12) and the project's audience and topic. The README's Python badge is a static `3.12+` so it renders regardless of the release's metadata.
|
|
18
|
+
|
|
19
|
+
### Deprecated
|
|
20
|
+
|
|
21
|
+
- The `AUTORESEARCH_*` environment names and `.autoresearch` paths are still accepted. The 0.1.0 notes said they would go in 0.1.1; a bug-fix release should not also break a deployment's environment, so their removal moves to 0.2.0.
|
|
22
|
+
|
|
23
|
+
## [0.1.0] - 2026-09-10
|
|
24
|
+
|
|
25
|
+
The first release of Outerloop — autoresearch agents that improve your benchmark
|
|
26
|
+
on your own code, your keys, and your compute. Built and used daily by the
|
|
27
|
+
Agentic Learning AI Lab at NYU.
|
|
28
|
+
|
|
29
|
+
### Highlights
|
|
30
|
+
|
|
31
|
+
- **The loop.** An agent proposes a change, runs the experiment on your cluster,
|
|
32
|
+
measures it against the base tree at the same seed, and opens a pull request
|
|
33
|
+
only when the benchmark actually improves. Every attempt gets a short report,
|
|
34
|
+
negative results included.
|
|
35
|
+
- **Verification first.** Outerloop re-measures every claim against the
|
|
36
|
+
repository's contract — it never trusts a number the agent reports — and a
|
|
37
|
+
reviewer panel reads the change and the claim before a pull request stands.
|
|
38
|
+
- **Your guardrails apply.** Agents cannot touch the benchmark, the budgets, or
|
|
39
|
+
your CI; your branch protection and required checks apply to them as to any
|
|
40
|
+
contributor. A pull request waits for a human by default; a repo can also let
|
|
41
|
+
clean ones merge themselves.
|
|
42
|
+
- **Runs where you do.** Slurm or a single GPU machine, behind one compute seam.
|
|
43
|
+
Sessions run in a scrubbed sandbox, and your model key and bot credentials
|
|
44
|
+
never leave the job.
|
|
45
|
+
- **Onboarding in a few commands.** `outerloop init` writes the config and sets
|
|
46
|
+
up auth (a per-adopter GitHub App, or a token), `outerloop start` runs the
|
|
47
|
+
loop, and `outerloop upgrade` moves a local install to the newest release. The
|
|
48
|
+
container image is pulled, not built.
|
|
49
|
+
- **Contracts and the board.** A repo declares its benchmark, scope, and budgets
|
|
50
|
+
in `.outerloop.yaml`; the climb board publishes every attempt and its outcome,
|
|
51
|
+
and the ledger keeps the honest record — refusals and negatives included.
|
|
52
|
+
- **Research lines and depth.** Each agent works its own line and reintegrates
|
|
53
|
+
the moving base; a run hibernates through long experiments and wakes to read
|
|
54
|
+
its results.
|
|
55
|
+
- **Advisory review and a weekly digest.** The reviewer posts structured,
|
|
56
|
+
advisory findings on every pull request, and a weekly maintainer scan files a
|
|
57
|
+
codebase-health digest.
|
|
58
|
+
|
|
59
|
+
The `AUTORESEARCH_*` environment names and `.autoresearch` paths from before the
|
|
60
|
+
rename are still accepted in 0.1; 0.1.1 kept them (see its notes) and 0.2.0
|
|
61
|
+
removes them. The full history since the first pre-release follows.
|
|
62
|
+
|
|
63
|
+
### Changed
|
|
64
|
+
|
|
65
|
+
- Install guide: a lab member who is not an org owner now has documented steps
|
|
66
|
+
for GitHub App auth — make the personal App public, request the install, an
|
|
67
|
+
org owner approves, then optionally transfer the App to the org — matching
|
|
68
|
+
what `init --github-app` prints on the personal-App fallback.
|
|
69
|
+
- Docs: corrected stale `autoresearch` command and module paths left by the
|
|
70
|
+
rename (onboarding, author-syscalls, architecture, role-cli), refreshed the
|
|
71
|
+
architecture component map (`budget` → `limits`, `report` →
|
|
72
|
+
`progress`/`climbboard`), and documented the tick scheduling knobs
|
|
73
|
+
`OUTERLOOP_CADENCE_MIN`, `OUTERLOOP_MAX_JOB_MINUTES`, and
|
|
74
|
+
`OUTERLOOP_MIN_TICK_MINUTES` in the install guide.
|
|
75
|
+
|
|
76
|
+
- The maintainer digest issue now carries the scan date in its title
|
|
77
|
+
(`Maintainer digest — YYYY-MM-DD`), so its freshness shows in the issue list.
|
|
78
|
+
The scan's long verdict and rejected-findings text now folds into a
|
|
79
|
+
collapsible block, keeping the top of the issue a short summary (title,
|
|
80
|
+
advisory line, counts).
|
|
81
|
+
|
|
82
|
+
### Fixed
|
|
83
|
+
|
|
84
|
+
- `outerloop init --github-app` no longer dead-ends a non-owner of an
|
|
85
|
+
organization. Creating an org-owned App needs org-owner rights, so GitHub
|
|
86
|
+
makes a personal App that will not install on the org repo by default. init
|
|
87
|
+
now detects that the App landed under a different account and prints the path
|
|
88
|
+
that works — make it public, request the install, have an org owner approve
|
|
89
|
+
it, then rerun `init --force --github-app` (and optionally transfer the App to
|
|
90
|
+
the org). The install-failure help names the same steps.
|
|
91
|
+
|
|
92
|
+
### Fixed
|
|
93
|
+
|
|
94
|
+
- The follow-up finalizer no longer gets stuck withholding a measured re-sync
|
|
95
|
+
when auto-merge is already off. Before pushing a re-measured head it confirms
|
|
96
|
+
the PR will not auto-merge by the PR's actual auto-merge state, rather than by
|
|
97
|
+
matching one error string — whose text varies with the repo's "Allow
|
|
98
|
+
auto-merge" setting and the token type — so a merge:manual repo under a GitHub
|
|
99
|
+
App token no longer re-withholds its successful re-measurements.
|
|
100
|
+
|
|
9
101
|
### Added
|
|
10
102
|
|
|
103
|
+
- `outerloop upgrade`: one verb for a local install to move to the newest
|
|
104
|
+
release (`--pre` to track pre-releases), the local counterpart of the Slurm
|
|
105
|
+
resident tick's `OUTERLOOP_AUTO_UPDATE` policy. It runs `pip install --upgrade`,
|
|
106
|
+
reports the version change, and reminds you to restart the loop.
|
|
107
|
+
- Base reintegration (docs/design/base-reintegration.md): when a research
|
|
108
|
+
line's base moves while it sleeps, its wake fetches the fresh base and tells
|
|
109
|
+
the agent to merge it and decide what to re-run — the same fetch-and-merge
|
|
110
|
+
the in-review conflict wake already uses, so a long depth run no longer
|
|
111
|
+
drifts behind sibling merges and opens a stale PR.
|
|
112
|
+
- `docs/design/base-reintegration.md`: a proposal to keep a running line's
|
|
113
|
+
base current by re-pinning at each wake and to reconcile a PR whose base
|
|
114
|
+
moved after the run ended, with the owner's five decisions and the build order.
|
|
11
115
|
- A weekly maintenance digest (`maintenance-agent.yml`, reusable; the kernel's
|
|
12
116
|
own caller is `maintenance.yml`): read-only lens sessions scan a repository's
|
|
13
117
|
default branch for dead pathways, duplicated logic, oversized modules, stale
|
|
@@ -649,6 +753,19 @@ Versions follow [SemVer](https://semver.org).
|
|
|
649
753
|
|
|
650
754
|
### Changed
|
|
651
755
|
|
|
756
|
+
- CI and reusable-workflow Actions bumped to current majors: actions/checkout
|
|
757
|
+
v7, astral-sh/setup-uv v10, upload-artifact v7, download-artifact v8. The
|
|
758
|
+
single-name artifact downloads stay flat and the pattern download reads
|
|
759
|
+
recursively, so the review split is unaffected; the node24 runtime is on the
|
|
760
|
+
hosted runners.
|
|
761
|
+
- The tick reads its run records once per phase instead of once per service: a
|
|
762
|
+
single snapshot after the mutation phase feeds the read services, and one
|
|
763
|
+
feeds the board pass. About twelve full run-directory scans per tick become
|
|
764
|
+
five, which matters under a large fleet (the 2026-09-03 disk pressure); the
|
|
765
|
+
freshness guards that re-read a single record are unchanged.
|
|
766
|
+
- The maintenance scan gains an `architecture` lens: it proposes abstraction
|
|
767
|
+
simplifications and missing extension points (a new backend, benchmark, or
|
|
768
|
+
role should need zero kernel change), as decision-kind digest items.
|
|
652
769
|
- Helpers shared across kernel modules are public in their owning module:
|
|
653
770
|
`brief.code_fence`, `brief.cap`, `attempt.target_clone_url`,
|
|
654
771
|
`attempt.stage_launch_job_ids`, `orchestrator.metric_from_output`,
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use Outerloop in your research, please cite it."
|
|
3
|
+
title: Outerloop
|
|
4
|
+
abstract: >-
|
|
5
|
+
Outerloop runs AI agents on your own research code: an agent proposes a
|
|
6
|
+
change, runs the experiment on your cluster, and opens a pull request only when
|
|
7
|
+
your benchmark improves. Every claim is re-measured against the repository's
|
|
8
|
+
contract, every attempt is recorded, and it runs on your keys, your compute,
|
|
9
|
+
and your repositories.
|
|
10
|
+
type: software
|
|
11
|
+
authors:
|
|
12
|
+
- name: "Agentic Learning AI Lab, New York University"
|
|
13
|
+
repository-code: "https://github.com/outerloop-science/outerloop"
|
|
14
|
+
url: "https://outerloop.science"
|
|
15
|
+
license: Apache-2.0
|
|
16
|
+
version: 0.1.1
|
|
17
|
+
date-released: "2026-09-11"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: outerloop-science
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.1
|
|
4
4
|
Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
|
|
5
5
|
Project-URL: Homepage, https://outerloop.science
|
|
6
6
|
Project-URL: Repository, https://github.com/outerloop-science/outerloop
|
|
@@ -10,6 +10,11 @@ Author: Agentic Learning AI Lab, New York University
|
|
|
10
10
|
License-Expression: Apache-2.0
|
|
11
11
|
License-File: LICENSE
|
|
12
12
|
License-File: NOTICE
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
18
|
Requires-Python: >=3.12
|
|
14
19
|
Requires-Dist: cryptography>=42
|
|
15
20
|
Requires-Dist: pydantic>=2
|
|
@@ -26,7 +31,7 @@ Description-Content-Type: text/markdown
|
|
|
26
31
|
|
|
27
32
|
[](https://github.com/outerloop-science/outerloop/actions/workflows/ci.yml)
|
|
28
33
|
[](https://pypi.org/project/outerloop-science/)
|
|
29
|
-
[](https://pypi.org/project/outerloop-science/)
|
|
30
35
|
[](LICENSE)
|
|
31
36
|
|
|
32
37
|
**Autoresearch agents that improve your benchmark.**
|
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
|
|
8
8
|
[](https://github.com/outerloop-science/outerloop/actions/workflows/ci.yml)
|
|
9
9
|
[](https://pypi.org/project/outerloop-science/)
|
|
10
|
-
[](https://pypi.org/project/outerloop-science/)
|
|
11
11
|
[](LICENSE)
|
|
12
12
|
|
|
13
13
|
**Autoresearch agents that improve your benchmark.**
|
|
@@ -8,8 +8,8 @@
|
|
|
8
8
|
|
|
9
9
|
## Cutting a release
|
|
10
10
|
|
|
11
|
-
1. PR: bump `__version__
|
|
12
|
-
version. A dev or rc pre-release still bumps the version (PyPI never
|
|
11
|
+
1. PR: bump `__version__`, set `CITATION.cff`'s `version` and `date-released`,
|
|
12
|
+
and move the `[Unreleased]` entries under the new version. A dev or rc pre-release still bumps the version (PyPI never
|
|
13
13
|
accepts a version twice, so the next one is `.dev1`, `rc2`, ...) but leaves
|
|
14
14
|
`[Unreleased]` in place until the final release.
|
|
15
15
|
2. `git tag vX.Y.Z && git push origin vX.Y.Z`. The `release` workflow builds
|
|
@@ -35,7 +35,7 @@ scope:
|
|
|
35
35
|
roadmap: docs/roadmap.md
|
|
36
36
|
```
|
|
37
37
|
|
|
38
|
-
The schema lives here (`
|
|
38
|
+
The schema lives here (`outerloop.contract`). The loader hard-codes invariants
|
|
39
39
|
no YAML can override: `.github/`, the contract file itself, and the roadmap are
|
|
40
40
|
always forbidden write paths; the contract is read from the default branch only;
|
|
41
41
|
**autoresearch is never a valid target of itself**.
|
|
@@ -278,7 +278,7 @@ insights across all targets and must never be part of any public flip. Raw
|
|
|
278
278
|
metrics stay in W&B; the notebook links to runs. No database or vector store —
|
|
279
279
|
grep + recency + distillation until that provably fails.
|
|
280
280
|
|
|
281
|
-
## Components (`src/
|
|
281
|
+
## Components (`src/outerloop/`)
|
|
282
282
|
|
|
283
283
|
| Module | Job |
|
|
284
284
|
| --- | --- |
|
|
@@ -287,8 +287,8 @@ grep + recency + distillation until that provably fails.
|
|
|
287
287
|
| `orchestrator` | Tick logic: sentinel, lease, task selection, session dispatch, state sync |
|
|
288
288
|
| `compute` | sbatch/squeue submit-and-poll behind one interface |
|
|
289
289
|
| `github` | Bot auth and push (orchestrator-side, after sessions end), PR/issue ops |
|
|
290
|
-
| `
|
|
291
|
-
| `
|
|
290
|
+
| `limits` | Effective session and job limits — turns, session and job minutes — with each contract wish clamped into our `[floor, ceiling]`; the tick's `OUTERLOOP_MAX_JOB_MINUTES` floors here |
|
|
291
|
+
| `progress` / `climbboard` | `progress` writes the human-readable benchmark-progress files into the target repo on each improvement PR; `climbboard` publishes the climb board — every ended attempt and its outcome, plus the ledger and views — to the target's `research-log` branch whenever runs end |
|
|
292
292
|
|
|
293
293
|
## Harness and context engineering
|
|
294
294
|
|
|
@@ -309,7 +309,7 @@ exactly what any given agent saw, and a bad run can be replayed from its brief:
|
|
|
309
309
|
| Ruler: how the metric is computed, how claims get re-verified | target repo docs | fixed |
|
|
310
310
|
| Lessons: distilled, bounded per-target lessons file | notebook `lessons/<target>.md` | hard cap |
|
|
311
311
|
| Recent history: last N run reports for this target (incl. failures) | notebook `runs/<target>/` | hard cap, newest first |
|
|
312
|
-
| Budget state: remaining GPU-hours/$/PRs this week |
|
|
312
|
+
| Budget state: remaining GPU-hours/$/PRs this week | tick weekly accounting | fixed |
|
|
313
313
|
|
|
314
314
|
Everything else is deliberately absent: no other targets' data (cross-target
|
|
315
315
|
separation), no raw transcripts (distillation instead), no maintainer-private
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
# Base reintegration: keeping a running line current, and reconciling its PR
|
|
2
|
+
|
|
3
|
+
Status: proposal (2026-09-08). Extends `research-lines.md`, which owns the
|
|
4
|
+
per-agent branch and the "merge main in at run start" rule. Prompted by a
|
|
5
|
+
real case, PR #12 on gpt-speedrun.
|
|
6
|
+
|
|
7
|
+
## The problem, concretely
|
|
8
|
+
|
|
9
|
+
agent-04's line started 2026-09-04 in the morning. That evening a sibling
|
|
10
|
+
merged #10, moving the record from 8192 to 7808. agent-04's line kept
|
|
11
|
+
iterating on its own 8192-era base for three more days and opened PR #12
|
|
12
|
+
claiming 7232, against a base that had been superseded on day one.
|
|
13
|
+
|
|
14
|
+
The claim was real: 7232 came from setting `warmdown_iters` to 6400, the same
|
|
15
|
+
knob #10 set to 4352, on a monotonic trend, and it holds against the current
|
|
16
|
+
base. But the PR conflicted, and by the time it opened, the run had
|
|
17
|
+
terminated, so nothing in the kernel could reconcile it. A human merged it by
|
|
18
|
+
hand.
|
|
19
|
+
|
|
20
|
+
Two gaps produced this:
|
|
21
|
+
|
|
22
|
+
- A line merges main only at run start. A long depth run therefore drifts
|
|
23
|
+
arbitrarily far from main as siblings merge, and `research-lines.md` already
|
|
24
|
+
names the danger ("a stale line reverting others' wins").
|
|
25
|
+
The kernel already reconciles a PR whose base moves AFTER it opens: an open PR
|
|
26
|
+
keeps its run in the in-review state, and the tick's follow-up conflict wake
|
|
27
|
+
(`followup.py`) fetches the moved base into the workspace and asks the agent to
|
|
28
|
+
merge and re-measure. So the only real gap is the first bullet — a line that
|
|
29
|
+
never re-syncs main mid-run and opens its PR against a base superseded days
|
|
30
|
+
earlier.
|
|
31
|
+
|
|
32
|
+
## Proposed protocol
|
|
33
|
+
|
|
34
|
+
1. **Re-pin at each wake, not only at run start.** When a line wakes for its
|
|
35
|
+
next depth iteration, the kernel merges main into the line and hands the
|
|
36
|
+
agent a short digest of what changed since it last ran — the sibling record
|
|
37
|
+
moves and the files they touched. The agent decides what to re-run given
|
|
38
|
+
the new base. This bounds drift to a single iteration instead of the whole
|
|
39
|
+
run.
|
|
40
|
+
|
|
41
|
+
2. **Never mid-experiment.** A running eval keeps its base. Re-pinning happens
|
|
42
|
+
only at the iteration boundary, the sleep-to-wake seam, so no in-flight
|
|
43
|
+
measurement is invalidated and no launch is wasted.
|
|
44
|
+
|
|
45
|
+
3. **Re-measure the baseline at the new merge-base.** This is the roadmap's
|
|
46
|
+
"baseline re-run at merge-base." `research-lines.md` already makes every
|
|
47
|
+
claim name the baseline pair it was measured against, so a re-pinned line
|
|
48
|
+
stays legible about which base each number used.
|
|
49
|
+
|
|
50
|
+
4. **Keep a PR reconcilable after the run's terminal.** A line whose PR is
|
|
51
|
+
open stays reachable until the PR is merged or closed, so a base that moves
|
|
52
|
+
after the PR opens fires the existing conflict wake — merge main,
|
|
53
|
+
re-measure, push — instead of stranding the PR for a human.
|
|
54
|
+
|
|
55
|
+
## Decisions (owner, 2026-09-08)
|
|
56
|
+
|
|
57
|
+
The semantic choices, settled — the mechanics follow from them.
|
|
58
|
+
|
|
59
|
+
1. **Re-pin cadence: only when main moved.** Each wake does a cheap
|
|
60
|
+
fetch-and-compare; it re-merges and re-pins only when a sibling actually
|
|
61
|
+
landed something. No churn when nothing changed.
|
|
62
|
+
2. **What the agent sees: the record move, plus each sibling PR's metric and a
|
|
63
|
+
one-line summary — not full diffs.** Enough to decide whether its line is
|
|
64
|
+
still worth pursuing, without flooding its context.
|
|
65
|
+
3. **A superseded axis: tell, don't force.** The wake digest says its axis was
|
|
66
|
+
beaten (e.g. "warmdown was taken further by #10"); the agent decides to
|
|
67
|
+
pivot or push on. The kernel never kills a line for it.
|
|
68
|
+
4. **Healing a PR after its base moves: already handled, no new mechanism.**
|
|
69
|
+
An open PR's run stays in-review, and the follow-up conflict wake already
|
|
70
|
+
fetches the moved base and asks the agent to merge and re-measure. The
|
|
71
|
+
note's earlier "post-terminal" framing was wrong (terra, #341): the run is
|
|
72
|
+
not terminated while its PR is open. Base reintegration is therefore stage 1
|
|
73
|
+
alone — closing the during-run drift; the post-open case needs nothing new.
|
|
74
|
+
5. **Who pays the re-measure GPU: only when the agent keeps its change.** The
|
|
75
|
+
re-measure is then a normal launch against the line's own budget. No eager
|
|
76
|
+
baseline re-runs on every merge.
|
|
77
|
+
|
|
78
|
+
**Build order.** The whole feature is one stage: re-pin at wake with the digest
|
|
79
|
+
(decisions 1–3), the re-measure paid only on keep (decision 5). Decision 4 needs
|
|
80
|
+
no code — the existing follow-up conflict wake already covers a base that moves
|
|
81
|
+
after the PR opens.
|
|
82
|
+
|
|
83
|
+
## Why this shape addresses the problem
|
|
84
|
+
|
|
85
|
+
The base pin is correct during a run — you cannot measure improvement against
|
|
86
|
+
a moving target. The failure is not the pin; it is holding one pin for a
|
|
87
|
+
multi-day run and having no reconciliation once the run ends. Waking the line to merge the fresh base keeps the stability the pin gives
|
|
88
|
+
while bounding the drift to a single iteration, and it reuses the exact
|
|
89
|
+
mechanism the in-review conflict wake already uses — the kernel fetches, the
|
|
90
|
+
agent merges — rather than adding new kernel git machinery. The post-open case
|
|
91
|
+
needs nothing more: an open PR's run stays in-review and that same conflict
|
|
92
|
+
wake already reconciles it.
|
|
@@ -15,7 +15,7 @@ honest when it says three steps:
|
|
|
15
15
|
|
|
16
16
|
## What the wizard does
|
|
17
17
|
|
|
18
|
-
`python -m
|
|
18
|
+
`python -m outerloop.init` (interactive; every answer has a flag for
|
|
19
19
|
non-interactive use):
|
|
20
20
|
|
|
21
21
|
- **Detect compute.** `sbatch` on PATH → Slurm mode (prompt for account and
|
|
@@ -47,7 +47,7 @@ non-interactive use):
|
|
|
47
47
|
The doctor is re-runnable on its own (`init --doctor`) and is the first
|
|
48
48
|
thing support asks for.
|
|
49
49
|
- **Print the start command.** Nothing starts implicitly. The command is
|
|
50
|
-
`
|
|
50
|
+
`outerloop start` on both paths: it submits the resident tick where
|
|
51
51
|
`sbatch` exists and runs the local loop elsewhere, taking the root and
|
|
52
52
|
placement from flags, the environment, or the `.env` the wizard wrote.
|
|
53
53
|
|
|
@@ -73,7 +73,7 @@ simultaneously the zero-Slurm on-ramp and the paper's Karpathy-loop
|
|
|
73
73
|
ablation cell (width 1 × serialized) — the same kernel, the compute seam
|
|
74
74
|
swapped, nothing else different.
|
|
75
75
|
|
|
76
|
-
- **The chain is a loop.** No sbatch → `
|
|
76
|
+
- **The chain is a loop.** No sbatch → `outerloop start` runs
|
|
77
77
|
`tick --loop`, a tick every cadence in the foreground. State stays in records on
|
|
78
78
|
disk, so killing and restarting the loop resumes exactly like the Slurm
|
|
79
79
|
chain surviving a dead tick.
|
|
@@ -98,8 +98,8 @@ swapped, nothing else different.
|
|
|
98
98
|
`OUTERLOOP_COMPUTE` selection at the two `SlurmCompute()` sites,
|
|
99
99
|
`tick --loop`, local-mode wake-arming skip, per-mode required-env
|
|
100
100
|
relaxation (no account/partition in local mode), and the one launch
|
|
101
|
-
command `
|
|
102
|
-
2. **The wizard**: `
|
|
101
|
+
command `outerloop start` (`src/outerloop/cli.py`). Done.
|
|
102
|
+
2. **The wizard**: `outerloop.init` — detection, prompts, `.env` writer,
|
|
103
103
|
doctor (PAT path first).
|
|
104
104
|
3. **The manifest flow**: App mint + installation discovery inside the
|
|
105
105
|
wizard.
|
|
@@ -252,9 +252,9 @@ cron (weekly) or by hand. It fans out one read-only lens session per digest
|
|
|
252
252
|
section (`maintain.MAINTENANCE_LENSES`, plus `general` for the whole
|
|
253
253
|
checklist) over a checkout of the calling repository's default branch, merges
|
|
254
254
|
the opinions with the same summarizer the wide review round uses, and posts
|
|
255
|
-
ONE rolling issue ("Maintainer digest"
|
|
256
|
-
|
|
257
|
-
notifies watchers. The role is `roles.maintainer_spec`: the reviewer's tools
|
|
255
|
+
ONE rolling issue (titled "Maintainer digest — <date>", the date of the digest
|
|
256
|
+
it currently shows) whose body each scan replaces; earlier digests stay in the
|
|
257
|
+
edit history, and a short comment per scan notifies watchers. The role is `roles.maintainer_spec`: the reviewer's tools
|
|
258
258
|
and verdict shape (findings through the syscall tool, no scope), a larger
|
|
259
259
|
budget because a tree is more to read than a diff. Items carry a section
|
|
260
260
|
(`--category`), an effort and risk estimate, and a kind: `change` is
|
|
@@ -30,7 +30,7 @@ The invariants, proven on #132/#133 and non-negotiable everywhere:
|
|
|
30
30
|
kernel-side (authoritative validators, PAT-out-of-session, scope, budgets).
|
|
31
31
|
The tool is ergonomics and fast feedback; authority stays in the kernel.
|
|
32
32
|
2. **Sandbox-side tools are standalone** (stdlib-only — the target repo has no
|
|
33
|
-
|
|
33
|
+
outerloop), which duplicates a little advisory validation, pinned by
|
|
34
34
|
parity tests (`test_artifact_path_check_matches_the_kernel`).
|
|
35
35
|
**Orchestrator-side roles** (reviewer/verifier/panel/planner/steward run on
|
|
36
36
|
our own checkout, not a target sandbox) import the real validators — no
|
|
@@ -105,8 +105,9 @@ always exits successfully so your PR stays green.
|
|
|
105
105
|
|
|
106
106
|
The same reviewer, pointed at your whole repository once a week: dead code,
|
|
107
107
|
duplicated logic, oversized modules, stale pins, slow tests, repeated work on
|
|
108
|
-
the hot path, documentation drift. It writes one issue, "Maintainer
|
|
109
|
-
|
|
108
|
+
the hot path, documentation drift. It writes one issue, titled "Maintainer
|
|
109
|
+
digest — <date>" (the date of the digest it shows), and replaces its body each
|
|
110
|
+
scan. It changes no code and opens no work orders.
|
|
110
111
|
Add `.github/workflows/maintenance.yml` with the same secret as the reviewer:
|
|
111
112
|
|
|
112
113
|
```yaml
|
|
@@ -227,6 +228,21 @@ declares Contents, Issues and Pull requests read-write and Metadata read,
|
|
|
227
228
|
nothing else. If the install step was cut short, run
|
|
228
229
|
`outerloop init --force --github-app` to finish and re-check it.
|
|
229
230
|
|
|
231
|
+
**If you are a member, not an owner, of the organization.** Creating an App
|
|
232
|
+
*owned by the org* needs org-owner rights, so init falls back to creating one
|
|
233
|
+
under your personal account — and a personal App does not install on an org
|
|
234
|
+
repo by default. You do not need an org-owned App or a shared key:
|
|
235
|
+
|
|
236
|
+
1. In the App's settings, make it **public** (a private App installs only on
|
|
237
|
+
its owner's account, which is why the org repo is not offered).
|
|
238
|
+
2. Request its installation on your repo; an **org owner approves** the request.
|
|
239
|
+
3. Run `outerloop init --force --github-app` to record the installation.
|
|
240
|
+
|
|
241
|
+
Your key never leaves your machine. To let the lab own the App centrally later,
|
|
242
|
+
transfer it to the org from the App's Advanced settings — the same key keeps
|
|
243
|
+
working, so nothing has to be recreated. init prints these steps itself when it
|
|
244
|
+
detects the personal-App fallback.
|
|
245
|
+
|
|
230
246
|
**A fine-grained PAT — the fallback.** For an org that already runs a machine
|
|
231
247
|
user, or one where you cannot create Apps: pick `pat` (or pass `--pat-file`).
|
|
232
248
|
Mint the token on the machine user with these settings:
|
|
@@ -284,8 +300,10 @@ release tag at the next cadence (pre-releases included; the repo is public, so
|
|
|
284
300
|
no credential is needed). `main` follows every merge and is meant for the
|
|
285
301
|
kernel's own developers. Whatever moves the checkout, you or the policy, the
|
|
286
302
|
environment is synced to the commit that is checked out, or the deploy rolls
|
|
287
|
-
back to the last commit whose environment was installed. The local loop runs
|
|
288
|
-
|
|
303
|
+
back to the last commit whose environment was installed. The local loop runs
|
|
304
|
+
the installed package, which has no such policy: run `outerloop upgrade` to move
|
|
305
|
+
it to the newest release (add `--pre` to track pre-releases), then start it
|
|
306
|
+
again. That is `pip install --upgrade outerloop-science` under one verb.
|
|
289
307
|
|
|
290
308
|
Experiments run wherever your `compute` backend says. Slurm is the first
|
|
291
309
|
backend; the interface is small (submit a job, poll for completion), so a CI
|
|
@@ -318,7 +336,14 @@ still read). A Codex author always runs contained, so it also needs the image
|
|
|
318
336
|
cluster, evals run inside the Apptainer image at `OUTERLOOP_IMAGE` (default
|
|
319
337
|
`~/outerloop-images/agent-py312.sif`) in a jail that binds only the
|
|
320
338
|
checked-out tree — an eval that needs data must fetch it into the tree, and
|
|
321
|
-
GPU jobs are requested per node (`--gpus-per-node`).
|
|
339
|
+
GPU jobs are requested per node (`--gpus-per-node`). The tick has three
|
|
340
|
+
scheduling knobs. `OUTERLOOP_CADENCE_MIN`, read from the `.env`, is how often the
|
|
341
|
+
chain ticks (minutes; default 30). Two finer ones are read from the tick's own
|
|
342
|
+
environment (set at launch, not the per-tick `.env`): `OUTERLOOP_MIN_TICK_MINUTES`
|
|
343
|
+
coalesces ticks that land too close together (0 disables; unset defaults to the
|
|
344
|
+
lesser of 10 minutes and half the cadence), and `OUTERLOOP_MAX_JOB_MINUTES` caps
|
|
345
|
+
the walltime the tick requests for the climb and author-sleep wake jobs it sizes
|
|
346
|
+
(clamped under a code ceiling).
|
|
322
347
|
|
|
323
348
|
**Local mode without an image.** On a machine with no Apptainer image,
|
|
324
349
|
`OUTERLOOP_COMPUTE=local` still runs. Sessions run under the harness's own
|
|
@@ -22,7 +22,7 @@ Then run one climb by hand (not via the tick):
|
|
|
22
22
|
|
|
23
23
|
```bash
|
|
24
24
|
source env.sh
|
|
25
|
-
uv run python -m
|
|
25
|
+
uv run python -m outerloop.attempt \
|
|
26
26
|
--target <org/repo> --benchmark <a-cheap-benchmark> \
|
|
27
27
|
--run-root <run-root> --image <image.sif> \
|
|
28
28
|
<the account/partition/limit args the tick normally passes>
|
|
@@ -11,6 +11,13 @@ license = "Apache-2.0"
|
|
|
11
11
|
license-files = ["LICENSE", "NOTICE"]
|
|
12
12
|
requires-python = ">=3.12"
|
|
13
13
|
authors = [{ name = "Agentic Learning AI Lab, New York University" }]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.12",
|
|
19
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
20
|
+
]
|
|
14
21
|
dependencies = [
|
|
15
22
|
"cryptography>=42", # GitHub App auth (RS256 JWTs), the recommended identity
|
|
16
23
|
"pydantic>=2",
|
|
@@ -17,8 +17,10 @@ network; the production signer lives behind the `app-auth` extra.
|
|
|
17
17
|
|
|
18
18
|
from __future__ import annotations
|
|
19
19
|
|
|
20
|
+
import argparse
|
|
20
21
|
import base64
|
|
21
22
|
import json
|
|
23
|
+
import os
|
|
22
24
|
import time
|
|
23
25
|
import urllib.error
|
|
24
26
|
import urllib.parse
|
|
@@ -29,6 +31,7 @@ from pathlib import Path
|
|
|
29
31
|
from typing import Any
|
|
30
32
|
|
|
31
33
|
from outerloop.github import AUTH_SAFE_OPENER, FileTokenProvider, TokenProvider
|
|
34
|
+
from outerloop.paths import CONFIG_DIR
|
|
32
35
|
|
|
33
36
|
# RS256-sign the JWT signing input, returning the raw signature bytes.
|
|
34
37
|
Signer = Callable[[bytes], bytes]
|
|
@@ -211,3 +214,17 @@ def resolve_bot_auth(pat_file: str | Path, app_file: str | Path = "") -> TokenPr
|
|
|
211
214
|
if str(app_file).strip():
|
|
212
215
|
return app_provider_from_file(Path(str(app_file)).expanduser())
|
|
213
216
|
return FileTokenProvider(Path(str(pat_file)).expanduser())
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def add_credential_args(parser: argparse.ArgumentParser) -> None:
|
|
220
|
+
"""Add the shared bot-auth options to a role parser: `--pat-file`, and
|
|
221
|
+
`--github-app-file`, which supplies installation tokens instead of the PAT
|
|
222
|
+
when set. `resolve_bot_auth` reads the pair. One owner so the defaults and
|
|
223
|
+
help cannot drift between attempt, followup, and steward."""
|
|
224
|
+
parser.add_argument("--pat-file", default=str(CONFIG_DIR / "bot_pat"))
|
|
225
|
+
parser.add_argument(
|
|
226
|
+
"--github-app-file",
|
|
227
|
+
default=os.environ.get("OUTERLOOP_GITHUB_APP_FILE", ""),
|
|
228
|
+
help="GitHub App config (JSON: app_id, installation_id, private_key); "
|
|
229
|
+
"when set, installation tokens replace the PAT",
|
|
230
|
+
)
|