outerloop-science 0.1.0.dev0__tar.gz → 0.1.0.dev1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/.gitignore +3 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/CHANGELOG.md +74 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/CLAUDE.md +1 -1
- outerloop_science-0.1.0.dev1/CONTRIBUTING.md +89 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/PKG-INFO +13 -9
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/README.md +11 -7
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/dispatcher.md +2 -1
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/onboarding.md +3 -2
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/reviewer-infra.md +24 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/install.md +18 -5
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/roadmap.md +3 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/pyproject.toml +8 -2
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/tick_deploy.sh +1 -1
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/__init__.py +1 -1
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/appmanifest.py +15 -10
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/attempt.py +29 -7
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/cli.py +24 -16
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/climbboard.py +209 -4
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/compute.py +27 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/followup.py +24 -7
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/init.py +192 -19
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/paths.py +12 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/tick.py +107 -37
- outerloop_science-0.1.0.dev1/tests/conftest.py +50 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_attempt.py +25 -12
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_climbboard.py +104 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_compute.py +24 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_followup.py +60 -0
- outerloop_science-0.1.0.dev1/tests/test_init.py +472 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_local_compute.py +3 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_start.py +48 -4
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_sweep_git_locks.py +7 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_tick.py +130 -0
- outerloop_science-0.1.0.dev1/tests/test_tiers.py +37 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/uv.lock +138 -0
- outerloop_science-0.1.0.dev0/CONTRIBUTING.md +0 -75
- outerloop_science-0.1.0.dev0/tests/test_init.py +0 -242
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/.pre-commit-config.yaml +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/.python-version +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/LICENSE +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/NOTICE +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/RELEASING.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/SECURITY.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/assets/icon-dark.svg +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/assets/icon-light.svg +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/assets/icon.svg +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/compute.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/contract.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/agent-substrate.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/architecture.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/consolidation.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/external.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/github-app-auth.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/headline.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/judge-placement.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/meta.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/orchestrator-verify.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/public-surface.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/research-lines.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/research-loop-buildout.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/research-loop.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/resident-tick.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/review-placement.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/role-cli.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/roles.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/design/scaling.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/reviewer.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/docs/validation/author-syscalls.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/examples/review.yml +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/README.md +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/install_codex.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/install_hermes.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/requeue_moved_successors.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/setup_branch_protection.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/sweep_git_locks.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/tick_chain.sbatch +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/scripts/tick_resident.sh +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/__main__.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/appauth.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/brief.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/contract.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/contract_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/disk.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/dispatch.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/github.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/harness.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/housekeeping.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/intake.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/limits.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/markers.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/measure.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/orchestrator.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/panel.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/posting.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/progress.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/py.typed +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/review.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/review_agent.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/review_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/review_post_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/review_summarize_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/role_runner.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/roles.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/rolespec.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/runstate.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/steward.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/style.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/syscall.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/syscall_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/verifier.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/verify_agent.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/verify_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/src/outerloop/verify_post_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_appauth.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_appmanifest.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_bot_aliases.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_bot_login.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_brief.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_channel_dir.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_codex_harness.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_contract.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_contract_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_contract_names.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_disk.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_dispatch.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_env_bridge.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_github.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_hardening.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_harness.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_hermes_harness.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_housekeeping.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_import.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_intake.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_limits.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_markers.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_measure.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_measure_and_decide.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_orchestrator.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_packaging.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_panel.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_paths.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_posting.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_progress.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_requeue_moved_successors.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review_agent.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review_hardening.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review_policy.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_review_summarize.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_role_runner.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_rolespec.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_runstate.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_steward.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_syscall.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_syscall_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_tick_chain_successors.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_tick_resident.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_verifier.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_verify_agent.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_verify_agent_cli.py +0 -0
- {outerloop_science-0.1.0.dev0 → outerloop_science-0.1.0.dev1}/tests/test_version.py +0 -0
|
@@ -6,6 +6,44 @@ Versions follow [SemVer](https://semver.org).
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
### Added
|
|
10
|
+
|
|
11
|
+
- `climb/status.json` carries the fleet's queue: the kernel's own Slurm jobs
|
|
12
|
+
(tick chain, sessions, wakes, evals, launches), each attributed to an agent,
|
|
13
|
+
with state, elapsed time, partition and submit time. Jobs on the account
|
|
14
|
+
that are not the kernel's never appear. The strip republishes when a job
|
|
15
|
+
appears, leaves, or changes state, not on elapsed drift.
|
|
16
|
+
- `outerloop init` asks for the author's model API key (hidden) and writes it to
|
|
17
|
+
`~/.config/outerloop/<backend>_key` (0600), or takes `--author-key-file` for an
|
|
18
|
+
existing file (checked to exist, stored absolute); the `.env` records it as
|
|
19
|
+
`OUTERLOOP_<BACKEND>_KEY_FILE`. The key lands in the config dir in use:
|
|
20
|
+
`~/.config/outerloop/`, or `~/.config/autoresearch/` on a machine set up
|
|
21
|
+
before the rename. The focused `--github-app` run still asks nothing about
|
|
22
|
+
the author. Every secret `init` writes (keys, PAT, App PEM and JSON, `.env`)
|
|
23
|
+
is now created 0600 in one step rather than written and then tightened.
|
|
24
|
+
|
|
25
|
+
### Changed
|
|
26
|
+
|
|
27
|
+
- The board's queue is a card in the live strip, not a preformatted dump: a
|
|
28
|
+
full-width table with running jobs first, state pills, elapsed time,
|
|
29
|
+
partition (first of several, the rest counted, all on hover), and the
|
|
30
|
+
owning agent in its color; long job names are cut with an ellipsis and
|
|
31
|
+
shown whole on hover.
|
|
32
|
+
- The Claude author's key follows the same rule as Codex's: the file is
|
|
33
|
+
`~/.config/outerloop/claude_key` and the setting `OUTERLOOP_CLAUDE_KEY_FILE`.
|
|
34
|
+
The pre-rename `harness_key` file and `*_HARNESS_KEY_FILE` setting are still
|
|
35
|
+
read (the new name wins when both exist), so an existing setup keeps working.
|
|
36
|
+
- `AUTORESEARCH_TARGET` is required for in-review servicing (follow-ups, wakes,
|
|
37
|
+
the board). The kernel no longer falls back to the retired
|
|
38
|
+
`agentic-learning-ai-lab/autoresearch-pilot` repo when it is unset; the tick
|
|
39
|
+
logs the missing setting like any other.
|
|
40
|
+
- The test suite runs on all cores by default (pytest-xdist), about four times
|
|
41
|
+
faster locally; `pytest --testmon` runs only the tests affected by your edits
|
|
42
|
+
(pytest-testmon). Tier selection moved from `addopts` into `tests/conftest.py`
|
|
43
|
+
(a `-m` in `addopts` disables testmon), and a `serial` tier holds the tests
|
|
44
|
+
that inspect the process table: skipped while workers run, and `pytest -m
|
|
45
|
+
serial` turns workers off, so the tier always runs alone (CI's second step).
|
|
46
|
+
|
|
9
47
|
### Changed
|
|
10
48
|
|
|
11
49
|
- The syscall channel dir in a workspace is now `.outerloop/` (the tool, the
|
|
@@ -51,6 +89,42 @@ Versions follow [SemVer](https://semver.org).
|
|
|
51
89
|
|
|
52
90
|
### Fixed
|
|
53
91
|
|
|
92
|
+
- `outerloop init --github-app` says what to do at each step: which page
|
|
93
|
+
opens, which button to click, where the code appears, how to install the
|
|
94
|
+
App on the repository, and that the last step checks write access. It also
|
|
95
|
+
asks whether to create the App under your account or an organization,
|
|
96
|
+
which before needed the undocumented `--org` flag.
|
|
97
|
+
- The App flow's final write check asked GitHub a question installation
|
|
98
|
+
tokens cannot answer, so it warned about a missing push permission on Apps
|
|
99
|
+
that could push. It now checks the installation's repository list and its
|
|
100
|
+
granted permissions.
|
|
101
|
+
- `init --github-app` records the App's login (`<slug>[bot]`) as
|
|
102
|
+
`OUTERLOOP_BOT_LOGIN`. Without it the kernel fell back to a built-in
|
|
103
|
+
default that is not the adopter's identity and did not recognize its own
|
|
104
|
+
pull requests.
|
|
105
|
+
- The local loop launches attempts without a container image. On a machine
|
|
106
|
+
with no Apptainer image, `AUTORESEARCH_COMPUTE=local` now brings up
|
|
107
|
+
servicing with an empty image, logs once that sessions run under the
|
|
108
|
+
harness's own sandbox and evaluations run bare on that machine, and passes
|
|
109
|
+
`--uncontained` to its jobs. The panel is off in that mode unless
|
|
110
|
+
`OUTERLOOP_PANEL_UNCONTAINED=1` opts in, and says so when it runs. Slurm
|
|
111
|
+
still requires the image, and so does a Codex author in every mode (#289).
|
|
112
|
+
- The PyPI description no longer refers to the lab.
|
|
113
|
+
- A base-synced re-measurement is finished instead of abandoned. The parked
|
|
114
|
+
stage recorded the session's local merge of the moved base as its parent,
|
|
115
|
+
a commit GitHub never saw, so the resume judged the PR head as "moved" on
|
|
116
|
+
every base-synced park and threw the landed result away. The stage now
|
|
117
|
+
records the PR head at park and the resume compares against that. An
|
|
118
|
+
abandon also releases the wake-attempt cap, so the "ask again" it invites
|
|
119
|
+
can be served.
|
|
120
|
+
- A follow-up re-measurement that has landed is now finished even when the
|
|
121
|
+
run has spent its wake-attempt cap. The sessions that sync a stale PR onto
|
|
122
|
+
the moved base and dispatch the measurement each bill an attempt, so a
|
|
123
|
+
measurement could complete with nobody left to push its number; the run then
|
|
124
|
+
idled with "follow-up attempts without progress" every tick. The finishing
|
|
125
|
+
follow-up has its own allowance of `MAX_WAKE_ATTEMPTS`, counted on the
|
|
126
|
+
stage, so a session that fails to finish is retried a bounded number of
|
|
127
|
+
times; the cap still holds for follow-ups that have nothing landed to finish.
|
|
54
128
|
- `outerloop init` no longer overwrites an existing `~/.config/autoresearch/.env`
|
|
55
129
|
silently: interactively it asks; with `--yes` it refuses unless `--force` is
|
|
56
130
|
passed. The check runs before any GitHub App is created.
|
|
@@ -8,7 +8,7 @@ Scaffold phase — `docs/roadmap.md` says what exists vs. planned;
|
|
|
8
8
|
|
|
9
9
|
```bash
|
|
10
10
|
uv sync
|
|
11
|
-
uv run pytest #
|
|
11
|
+
uv run pytest # all cores; --testmon = only tests affected by your edits
|
|
12
12
|
uv run ruff check --fix . && uv run ruff format .
|
|
13
13
|
uv run mypy
|
|
14
14
|
uv run pre-commit run --all-files
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Outerloop is developed in the open by the [Agentic Learning AI Lab](https://agenticlearning.ai)
|
|
4
|
+
at NYU, and its own agents are regular contributors to it. Contributions from
|
|
5
|
+
people are welcome too: bug reports, fixes, documentation, new compute
|
|
6
|
+
backends, and benchmark contracts for your own repos.
|
|
7
|
+
|
|
8
|
+
## Before you start
|
|
9
|
+
|
|
10
|
+
- Look through the open issues. For anything larger than a small fix, open an
|
|
11
|
+
issue first so we can agree on the shape before you write code.
|
|
12
|
+
- Security problems go to the contact in [SECURITY.md](SECURITY.md), not to a
|
|
13
|
+
public issue.
|
|
14
|
+
|
|
15
|
+
## Setup
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
uv sync
|
|
19
|
+
uv run pre-commit install
|
|
20
|
+
uv run pytest
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
## Making a change
|
|
24
|
+
|
|
25
|
+
1. Branch from `main`: `fix/<topic>` or `feat/<topic>`.
|
|
26
|
+
2. Keep the change focused. Add or update tests, and a line under
|
|
27
|
+
`[Unreleased]` in `CHANGELOG.md` for anything a user would notice.
|
|
28
|
+
3. Run the same gate CI runs before you push:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
uv run pre-commit run --all-files # lint, formatting, secret scan
|
|
32
|
+
uv lock --check
|
|
33
|
+
uv run mypy
|
|
34
|
+
uv run pytest && uv run pytest -m serial
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
4. Open a pull request against `main`. Say what changed, why, and how you
|
|
38
|
+
verified it.
|
|
39
|
+
|
|
40
|
+
## What happens to your pull request
|
|
41
|
+
|
|
42
|
+
- CI must pass.
|
|
43
|
+
- One of our agents reviews it and posts its findings as a review, usually
|
|
44
|
+
within the hour; no time is guaranteed. It never approves or blocks; a
|
|
45
|
+
maintainer decides. Fix what is right and reply to what is not; findings
|
|
46
|
+
answered with a reason are fine. After you push fixes, a maintainer removes
|
|
47
|
+
and re-adds the `outerloop:review` label to run another round.
|
|
48
|
+
- The agent review runs only for branches in this repository. A pull request
|
|
49
|
+
from a fork gets CI but no agent review; a maintainer can push your branch
|
|
50
|
+
here to run one.
|
|
51
|
+
- A maintainer merges with a merge commit. We do not squash or rebase, so the
|
|
52
|
+
history of how a change came to be stays intact. Keep your branch current
|
|
53
|
+
with `git merge main`, and never force-push a branch someone else has seen.
|
|
54
|
+
|
|
55
|
+
## Conventions
|
|
56
|
+
|
|
57
|
+
- Python 3.12, absolute imports (`from outerloop ...`), and code that works
|
|
58
|
+
from the installed wheel; no paths relative to a checkout.
|
|
59
|
+
- `ruff` for lint and formatting (line length 100); `mypy` clean.
|
|
60
|
+
- Dependencies go in the `pyproject.toml` group that owns them, then
|
|
61
|
+
`uv lock`; CI checks the lock.
|
|
62
|
+
- Docs are plain prose. Short sentences, no metaphors, no internal jargon.
|
|
63
|
+
|
|
64
|
+
## Tests
|
|
65
|
+
|
|
66
|
+
| Tier | Marker | Runs |
|
|
67
|
+
| --- | --- | --- |
|
|
68
|
+
| unit | *(none)* | CI, every PR, on all cores |
|
|
69
|
+
| serial | `serial` | CI, second step: `pytest -m serial` (inspects the process table) |
|
|
70
|
+
| slow | `slow` | nightly or manual |
|
|
71
|
+
| llm | `llm` | manual only, paid APIs |
|
|
72
|
+
| slurm | `slurm` | manual only, needs a cluster |
|
|
73
|
+
|
|
74
|
+
While editing, `uv run pytest --testmon` runs only the tests affected by your
|
|
75
|
+
change; `uv run pytest -n0` runs everything serially.
|
|
76
|
+
|
|
77
|
+
## Pull requests from the agents
|
|
78
|
+
|
|
79
|
+
Pull requests authored by `outerloop-science[bot]` are the system improving
|
|
80
|
+
the repositories it works on. They appear on those target repositories and are
|
|
81
|
+
skipped by the advisory reviewer by design. What gates them is each target's
|
|
82
|
+
own rules: its required human review, or, where a target's contract allows
|
|
83
|
+
automatic merging, the measurement gate and review panel that ran before the
|
|
84
|
+
PR opened. This document is about contributing to Outerloop, this repository.
|
|
85
|
+
|
|
86
|
+
## License
|
|
87
|
+
|
|
88
|
+
By contributing you agree that your contributions are licensed under the
|
|
89
|
+
[Apache License 2.0](LICENSE), the same as the project.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: outerloop-science
|
|
3
|
-
Version: 0.1.0.
|
|
4
|
-
Summary: Autonomous research
|
|
3
|
+
Version: 0.1.0.dev1
|
|
4
|
+
Summary: Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time
|
|
5
5
|
Project-URL: Homepage, https://outerloop.science
|
|
6
6
|
Project-URL: Repository, https://github.com/outerloop-science/outerloop
|
|
7
7
|
Project-URL: Changelog, https://github.com/outerloop-science/outerloop/blob/main/CHANGELOG.md
|
|
@@ -56,18 +56,22 @@ themselves.
|
|
|
56
56
|
|
|
57
57
|
## Get started
|
|
58
58
|
|
|
59
|
-
Three commands and
|
|
60
|
-
|
|
61
|
-
|
|
59
|
+
Three commands and one file. You need a repo with a benchmark command, an API
|
|
60
|
+
key for the model that will write the code, and a Slurm cluster or one machine
|
|
61
|
+
with a GPU.
|
|
62
62
|
|
|
63
63
|
```bash
|
|
64
64
|
pip install outerloop-science
|
|
65
|
-
outerloop init # where the loop runs, which repo,
|
|
65
|
+
outerloop init # where the loop runs, which repo, which model and its key, your GitHub identity
|
|
66
66
|
```
|
|
67
67
|
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
68
|
+
The wizard asks for a GitHub identity for the agents to open pull requests
|
|
69
|
+
as. Pick `app` and it walks you through creating a GitHub App, under your
|
|
70
|
+
account or under an organization you name, and installing it on the repo:
|
|
71
|
+
two browser pages and a code pasted back. It then checks that the App can
|
|
72
|
+
write the repo and tells you if it cannot. Pick `pat` if you already have a
|
|
73
|
+
token. It writes the config and the key files; nothing to edit by hand. Then
|
|
74
|
+
add one file, `.outerloop.yaml`, to the repo you want improved:
|
|
71
75
|
|
|
72
76
|
```yaml
|
|
73
77
|
benchmarks:
|
|
@@ -37,18 +37,22 @@ themselves.
|
|
|
37
37
|
|
|
38
38
|
## Get started
|
|
39
39
|
|
|
40
|
-
Three commands and
|
|
41
|
-
|
|
42
|
-
|
|
40
|
+
Three commands and one file. You need a repo with a benchmark command, an API
|
|
41
|
+
key for the model that will write the code, and a Slurm cluster or one machine
|
|
42
|
+
with a GPU.
|
|
43
43
|
|
|
44
44
|
```bash
|
|
45
45
|
pip install outerloop-science
|
|
46
|
-
outerloop init # where the loop runs, which repo,
|
|
46
|
+
outerloop init # where the loop runs, which repo, which model and its key, your GitHub identity
|
|
47
47
|
```
|
|
48
48
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
49
|
+
The wizard asks for a GitHub identity for the agents to open pull requests
|
|
50
|
+
as. Pick `app` and it walks you through creating a GitHub App, under your
|
|
51
|
+
account or under an organization you name, and installing it on the repo:
|
|
52
|
+
two browser pages and a code pasted back. It then checks that the App can
|
|
53
|
+
write the repo and tells you if it cannot. Pick `pat` if you already have a
|
|
54
|
+
token. It writes the config and the key files; nothing to edit by hand. Then
|
|
55
|
+
add one file, `.outerloop.yaml`, to the repo you want improved:
|
|
52
56
|
|
|
53
57
|
```yaml
|
|
54
58
|
benchmarks:
|
|
@@ -236,7 +236,8 @@ which wakes any run whose job reads terminal after the grace period —
|
|
|
236
236
|
plus GONE past the deadline (vanished) and PENDING past the deadline
|
|
237
237
|
(cancel, then wake as unschedulable); a still-RUNNING job is deliberately
|
|
238
238
|
left alone, bounded by its own walltime; wakes that fire without producing
|
|
239
|
-
progress -> `stuck` at MAX_WAKE_ATTEMPTS
|
|
239
|
+
progress -> `stuck` at MAX_WAKE_ATTEMPTS (a landed re-measure has its own
|
|
240
|
+
allowance of MAX_WAKE_ATTEMPTS finishing follow-ups, counted on the stage). A moved base during the wait is
|
|
240
241
|
review's to handle — the sealed candidate publishes as-is (research-loop.md:
|
|
241
242
|
a stale PR is a re-wake, never an orchestrator auto-merge). Worktree cleanup has a named owner at every exit: the
|
|
242
243
|
wake's own finally (primary, as in-job measures do today), the sweep's
|
|
@@ -34,8 +34,9 @@ non-interactive use):
|
|
|
34
34
|
discovers the installation id itself (App JWT → `GET /app/installations`).
|
|
35
35
|
A pre-existing PAT path is accepted as the alternative for orgs that
|
|
36
36
|
prefer it.
|
|
37
|
-
- **Collect the author key.**
|
|
38
|
-
|
|
37
|
+
- **Collect the author key.** A hidden paste for the configured backend,
|
|
38
|
+
written to `~/.config/outerloop/<backend>_key` (0600) and recorded as
|
|
39
|
+
`OUTERLOOP_<BACKEND>_KEY_FILE`; `--author-key-file` for an existing file.
|
|
39
40
|
- **Write `~/.config/outerloop/.env`** — only keys the operator chose;
|
|
40
41
|
the tick chain's allowlist is the contract for what matters.
|
|
41
42
|
- **Doctor.** Re-run every check the tick already preflights, plus the
|
|
@@ -334,3 +334,27 @@ the same infrastructure the fork-PR phase needs.
|
|
|
334
334
|
Re-test on version bumps: whether Claude's `Read` opens `/proc`, a hermes
|
|
335
335
|
read-only toolset, a codex sandbox that hides `/proc`, or a runner that grants
|
|
336
336
|
PID namespaces would each rewrite this section.
|
|
337
|
+
|
|
338
|
+
## Maintainers: how many review rounds
|
|
339
|
+
|
|
340
|
+
The advisory reviewer runs on open; a maintainer re-runs it after a fix by
|
|
341
|
+
removing and re-adding the `outerloop:review` label once the push has
|
|
342
|
+
settled (the workflow fires on the labeled event). Only the most recent round,
|
|
343
|
+
run against the head commit, counts; a quiet verdict on an older diff
|
|
344
|
+
authorizes nothing. Findings rejected on rationale get a reply on the thread.
|
|
345
|
+
|
|
346
|
+
Termination is judged, not literal, because an eager reviewer can always find
|
|
347
|
+
one more wording nit:
|
|
348
|
+
|
|
349
|
+
- code PRs iterate until a round yields no new medium-or-higher or
|
|
350
|
+
behavior-affecting finding; wording nits in an otherwise quiet round are
|
|
351
|
+
fixed or answered without another round;
|
|
352
|
+
- docs and process PRs get one round, its nits batched into one fix;
|
|
353
|
+
- hard cap of four rounds: still finding mediums by then means the change is
|
|
354
|
+
the problem; stop and rethink rather than cycle.
|
|
355
|
+
|
|
356
|
+
The habit exists because later rounds have repeatedly found real defects in
|
|
357
|
+
earlier rounds' own fixes. Bot-authored improvement PRs sit outside this gate:
|
|
358
|
+
the reviewer skips them by design, and their gate is the target repo's
|
|
359
|
+
required human review (the publish step arms auto-merge only when the target's
|
|
360
|
+
branch protection requires one).
|
|
@@ -14,7 +14,7 @@ minutes and needs no bot account, no cluster, and no GPU. Do that first.
|
|
|
14
14
|
An automated reviewer comments on your pull requests. It never approves, never
|
|
15
15
|
blocks a merge, and never fails your build.
|
|
16
16
|
|
|
17
|
-
**You need:** an
|
|
17
|
+
**You need:** an API key for the reviewer's model. Anthropic by default; the OpenAI and OpenRouter variants are below. That's it.
|
|
18
18
|
|
|
19
19
|
**Step 1 — add the key as a repository secret.** Repo → Settings → Secrets and
|
|
20
20
|
variables → Actions → New repository secret. Name it `ANTHROPIC_REVIEWER_KEY`.
|
|
@@ -187,7 +187,7 @@ add it to a team — teams grant more than it needs and inherit future grants.
|
|
|
187
187
|
The quickest path is the guided setup:
|
|
188
188
|
|
|
189
189
|
```bash
|
|
190
|
-
outerloop init # asks for compute, target repo, placement, and
|
|
190
|
+
outerloop init # asks for compute, target repo, placement, auth, and the author's model key
|
|
191
191
|
outerloop start
|
|
192
192
|
```
|
|
193
193
|
|
|
@@ -226,13 +226,26 @@ comma-separated partition list lets Slurm start each job wherever it fits
|
|
|
226
226
|
first; `OUTERLOOP_PANEL` names the verify/review lenses (with
|
|
227
227
|
`OUTERLOOP_PANEL_*_KEY_FILE` for their keys); the author backend is
|
|
228
228
|
`OUTERLOOP_AUTHOR_BACKEND`/`OUTERLOOP_AUTHOR_MODEL`, its key file
|
|
229
|
-
`
|
|
230
|
-
|
|
231
|
-
|
|
229
|
+
`OUTERLOOP_<BACKEND>_KEY_FILE` (`OUTERLOOP_CLAUDE_KEY_FILE`,
|
|
230
|
+
`OUTERLOOP_CODEX_KEY_FILE`; `init` writes the key to
|
|
231
|
+
`~/.config/outerloop/<backend>_key`, 0600, and a pre-rename `harness_key` is
|
|
232
|
+
still read). A Codex author always runs contained, so it also needs the image
|
|
233
|
+
(`OUTERLOOP_IMAGE`) and a Codex model in `OUTERLOOP_AUTHOR_MODEL`. On a
|
|
234
|
+
cluster, evals run inside the Apptainer image at `OUTERLOOP_IMAGE` (default
|
|
232
235
|
`~/autoresearch-images/agent-py312.sif`) in a jail that binds only the
|
|
233
236
|
checked-out tree — an eval that needs data must fetch it into the tree, and
|
|
234
237
|
GPU jobs are requested per node (`--gpus-per-node`).
|
|
235
238
|
|
|
239
|
+
**Local mode without an image.** On a machine with no Apptainer image,
|
|
240
|
+
`OUTERLOOP_COMPUTE=local` still runs. Sessions run under the harness's own
|
|
241
|
+
sandbox, evaluations run bare in a throwaway tree under an allowlisted
|
|
242
|
+
environment, both on your machine with your keys, and the loop says so once
|
|
243
|
+
at start. The verification panel is off in this mode unless you set
|
|
244
|
+
`OUTERLOOP_PANEL_UNCONTAINED=1`, because an uncontained judge holds a shell
|
|
245
|
+
next to its own key file; a pull request opened without a panel says so. A
|
|
246
|
+
Codex author needs the image in every mode. Contained local mode, the
|
|
247
|
+
default on Linux once the image is published, needs Apptainer and the image.
|
|
248
|
+
|
|
236
249
|
---
|
|
237
250
|
|
|
238
251
|
## Billing Claude sessions to GCP credits (Vertex AI)
|
|
@@ -100,6 +100,9 @@ standing instruction they supersede — a resumed agent honors stale constraints
|
|
|
100
100
|
|
|
101
101
|
## Phase 5 — Benchmark-climb pilot
|
|
102
102
|
|
|
103
|
+
Retired 2026-09-06: the fleet has climbed gpt-speedrun since 2026-08-31 and the
|
|
104
|
+
pilot repo is archived. Kept for the record.
|
|
105
|
+
|
|
103
106
|
Target: [autoresearch-pilot](https://github.com/agentic-learning-ai-lab/autoresearch-pilot)
|
|
104
107
|
— a non-research-bearing proving ground (tsp / denoise / speedup; deterministic,
|
|
105
108
|
CPU-only, contract and baselines already committed). Decouples this roadmap from
|
|
@@ -5,7 +5,7 @@ build-backend = "hatchling.build"
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "outerloop-science"
|
|
7
7
|
dynamic = ["version"]
|
|
8
|
-
description = "Autonomous research
|
|
8
|
+
description = "Autonomous research agents that improve the benchmark you point them at, one verified pull request at a time"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "Apache-2.0"
|
|
11
11
|
license-files = ["LICENSE", "NOTICE"]
|
|
@@ -35,6 +35,8 @@ dev = [
|
|
|
35
35
|
"mypy>=1.13",
|
|
36
36
|
"pytest>=8",
|
|
37
37
|
"pre-commit>=4",
|
|
38
|
+
"pytest-xdist>=3",
|
|
39
|
+
"pytest-testmon>=2",
|
|
38
40
|
]
|
|
39
41
|
|
|
40
42
|
[project.urls]
|
|
@@ -68,11 +70,15 @@ warn_unused_ignores = true
|
|
|
68
70
|
|
|
69
71
|
[tool.pytest.ini_options]
|
|
70
72
|
testpaths = ["tests"]
|
|
71
|
-
|
|
73
|
+
# parallel by default; tests/conftest.py applies the tiers (slow/llm/slurm off,
|
|
74
|
+
# serial off while distributing). `-n0` runs serially, `--testmon` runs only
|
|
75
|
+
# the tests affected by your edits.
|
|
76
|
+
addopts = "-ra --strict-markers -n auto"
|
|
72
77
|
markers = [
|
|
73
78
|
"slow: long-running; nightly/manual only",
|
|
74
79
|
"llm: calls a paid LLM API; manual only, never CI",
|
|
75
80
|
"slurm: needs a Slurm cluster; manual only, never CI",
|
|
81
|
+
"serial: inspects the process table; runs alone via `pytest -n0 -m serial`",
|
|
76
82
|
]
|
|
77
83
|
|
|
78
84
|
[tool.hatch.build.targets.sdist]
|
|
@@ -96,7 +96,7 @@ if [ -r "$ENV_FILE" ]; then
|
|
|
96
96
|
if [ "$owner" = "$(id -u)" ] && [ $((8#$perms & 8#022)) -eq 0 ]; then
|
|
97
97
|
for _k in AUTORESEARCH_AUTHOR_BACKEND AUTORESEARCH_AUTHOR_MODEL \
|
|
98
98
|
AUTORESEARCH_CODEX_BIN AUTORESEARCH_CODEX_KEY_FILE \
|
|
99
|
-
AUTORESEARCH_HARNESS_KEY_FILE \
|
|
99
|
+
AUTORESEARCH_CLAUDE_KEY_FILE AUTORESEARCH_HARNESS_KEY_FILE \
|
|
100
100
|
AUTORESEARCH_VERTEX_PROJECT AUTORESEARCH_VERTEX_REGION \
|
|
101
101
|
AUTORESEARCH_VERTEX_ADC \
|
|
102
102
|
AUTORESEARCH_TARGET \
|
|
@@ -25,12 +25,14 @@ import os
|
|
|
25
25
|
import secrets
|
|
26
26
|
import time
|
|
27
27
|
import urllib.error
|
|
28
|
+
import urllib.parse
|
|
28
29
|
import urllib.request
|
|
29
30
|
from collections.abc import Callable
|
|
30
31
|
from pathlib import Path
|
|
31
32
|
from typing import Any
|
|
32
33
|
|
|
33
34
|
from outerloop.appauth import API, build_app_jwt, signer_from_private_key
|
|
35
|
+
from outerloop.paths import write_private
|
|
34
36
|
|
|
35
37
|
# The hosted helper page: it auto-POSTs the manifest to GitHub, then displays the
|
|
36
38
|
# returned code. It is also the manifest's redirect_url, so GitHub sends the code
|
|
@@ -90,10 +92,15 @@ def request_manifest_code(
|
|
|
90
92
|
state = secrets.token_urlsafe(16)
|
|
91
93
|
manifest = build_manifest(name, homepage_url)
|
|
92
94
|
url = build_setup_url(manifest, state, org)
|
|
93
|
-
|
|
95
|
+
where = f"the {org} organization" if org else "your account"
|
|
96
|
+
print_fn("Step 1 of 3: create the GitHub App.")
|
|
97
|
+
print_fn(" Open this link in a browser where you are signed in to GitHub (any machine):")
|
|
94
98
|
print_fn(f"\n {url}\n")
|
|
95
|
-
print_fn("
|
|
96
|
-
|
|
99
|
+
print_fn(f" GitHub shows a page titled 'Create GitHub App' under {where}. Click the")
|
|
100
|
+
host = urllib.parse.urlsplit(SETUP_URL).netloc or SETUP_URL
|
|
101
|
+
print_fn(" green 'Create GitHub App' button. GitHub then sends you to")
|
|
102
|
+
print_fn(f" {host}, which shows a one-time code.")
|
|
103
|
+
return input_fn("Paste that code here: ").strip()
|
|
97
104
|
|
|
98
105
|
|
|
99
106
|
def _default_transport(request: urllib.request.Request) -> Any:
|
|
@@ -136,10 +143,10 @@ def save_app_creds(
|
|
|
136
143
|
config_dir.mkdir(parents=True, exist_ok=True)
|
|
137
144
|
slug = str(conversion["slug"])
|
|
138
145
|
pem_path = config_dir / f"{slug}-app.pem"
|
|
139
|
-
pem_path
|
|
140
|
-
pem_path.chmod(0o600)
|
|
146
|
+
write_private(pem_path, str(conversion["pem"]))
|
|
141
147
|
app_json = config_dir / f"github_app.{slug}.json"
|
|
142
|
-
|
|
148
|
+
write_private(
|
|
149
|
+
app_json,
|
|
143
150
|
json.dumps(
|
|
144
151
|
{
|
|
145
152
|
"app_id": int(conversion["id"]),
|
|
@@ -148,9 +155,8 @@ def save_app_creds(
|
|
|
148
155
|
},
|
|
149
156
|
indent=2,
|
|
150
157
|
)
|
|
151
|
-
+ "\n"
|
|
158
|
+
+ "\n",
|
|
152
159
|
)
|
|
153
|
-
app_json.chmod(0o600)
|
|
154
160
|
return pem_path, app_json
|
|
155
161
|
|
|
156
162
|
|
|
@@ -194,5 +200,4 @@ def set_installation_id(app_json: Path, installation_id: int) -> None:
|
|
|
194
200
|
"""Fill the installation id into an already-written github_app.<slug>.json."""
|
|
195
201
|
data = json.loads(app_json.read_text())
|
|
196
202
|
data["installation_id"] = int(installation_id)
|
|
197
|
-
app_json
|
|
198
|
-
app_json.chmod(0o600)
|
|
203
|
+
write_private(app_json, json.dumps(data, indent=2) + "\n")
|
|
@@ -97,8 +97,11 @@ log = logging.getLogger(__name__)
|
|
|
97
97
|
# tick preflights the panel key (and compares it against the author key —
|
|
98
98
|
# role separation) before claiming/submitting.
|
|
99
99
|
PANEL_KEY_DEFAULT = str(CONFIG_DIR / "verifier_key")
|
|
100
|
-
|
|
101
|
-
|
|
100
|
+
# One rule for author keys: ~/.config/outerloop/<backend>_key. `harness_key` is
|
|
101
|
+
# the pre-rename name of the claude key; it is still read, never written.
|
|
102
|
+
CLAUDE_KEY_DEFAULT = str(CONFIG_DIR / "claude_key")
|
|
103
|
+
HARNESS_KEY_DEFAULT = str(CONFIG_DIR / "harness_key") # legacy claude key
|
|
104
|
+
CODEX_KEY_DEFAULT = str(CONFIG_DIR / "codex_key")
|
|
102
105
|
|
|
103
106
|
|
|
104
107
|
def resolve_author_key_file(backend: str, explicit: str = "") -> str:
|
|
@@ -106,14 +109,27 @@ def resolve_author_key_file(backend: str, explicit: str = "") -> str:
|
|
|
106
109
|
codex's both on disk), selected by backend — so the author backend is a
|
|
107
110
|
config choice, not a key swap, and an in-flight run of either backend can
|
|
108
111
|
still be woken/serviced after a fleet flip. An explicit path always wins;
|
|
109
|
-
otherwise the per-backend env var, then the
|
|
110
|
-
|
|
111
|
-
|
|
112
|
+
otherwise the per-backend env var, then the default path. For claude the
|
|
113
|
+
legacy spellings (`AUTORESEARCH_HARNESS_KEY_FILE`, `harness_key`) are still
|
|
114
|
+
honored, so a machine set up before the rename keeps working; the new name
|
|
115
|
+
wins when both exist. The result is always ~-expanded, so every caller gets
|
|
116
|
+
a real path (an env value like "~/.config/..." must not reach the token
|
|
117
|
+
provider verbatim)."""
|
|
112
118
|
if not explicit:
|
|
113
119
|
if backend == "codex":
|
|
114
120
|
explicit = os.environ.get("AUTORESEARCH_CODEX_KEY_FILE") or CODEX_KEY_DEFAULT
|
|
115
121
|
else:
|
|
116
|
-
explicit =
|
|
122
|
+
explicit = (
|
|
123
|
+
os.environ.get("AUTORESEARCH_CLAUDE_KEY_FILE")
|
|
124
|
+
or os.environ.get("AUTORESEARCH_HARNESS_KEY_FILE")
|
|
125
|
+
or ""
|
|
126
|
+
)
|
|
127
|
+
if not explicit:
|
|
128
|
+
explicit = CLAUDE_KEY_DEFAULT
|
|
129
|
+
if not os.path.exists(os.path.expanduser(explicit)) and os.path.exists(
|
|
130
|
+
os.path.expanduser(HARNESS_KEY_DEFAULT)
|
|
131
|
+
):
|
|
132
|
+
explicit = HARNESS_KEY_DEFAULT
|
|
117
133
|
return os.path.expanduser(explicit)
|
|
118
134
|
|
|
119
135
|
|
|
@@ -2179,6 +2195,12 @@ def _panel_lenses_from_args(args: Any) -> tuple[tuple[PanelLens, ...], tuple[str
|
|
|
2179
2195
|
else:
|
|
2180
2196
|
lens_key = panel_key
|
|
2181
2197
|
try:
|
|
2198
|
+
if not args.image:
|
|
2199
|
+
log.warning(
|
|
2200
|
+
"panel lens %s runs uncontained (local mode, no image): the judge "
|
|
2201
|
+
"shares this machine with the operator's keys",
|
|
2202
|
+
backend,
|
|
2203
|
+
)
|
|
2182
2204
|
judge = build_harness(
|
|
2183
2205
|
lens_key,
|
|
2184
2206
|
reviewer_spec(),
|
|
@@ -3211,7 +3233,7 @@ def main() -> int:
|
|
|
3211
3233
|
"--key-file",
|
|
3212
3234
|
default="",
|
|
3213
3235
|
help="author key file; default resolves per backend (config-driven): "
|
|
3214
|
-
"
|
|
3236
|
+
"AUTORESEARCH_CLAUDE_KEY_FILE for claude, AUTORESEARCH_CODEX_KEY_FILE for codex",
|
|
3215
3237
|
)
|
|
3216
3238
|
parser.add_argument("--issue", type=int, default=0)
|
|
3217
3239
|
parser.add_argument(
|
|
@@ -46,6 +46,7 @@ TICK_ENV_KEYS = (
|
|
|
46
46
|
"AUTORESEARCH_AUTHOR_MODEL",
|
|
47
47
|
"AUTORESEARCH_CODEX_BIN",
|
|
48
48
|
"AUTORESEARCH_CODEX_KEY_FILE",
|
|
49
|
+
"AUTORESEARCH_CLAUDE_KEY_FILE",
|
|
49
50
|
"AUTORESEARCH_HARNESS_KEY_FILE",
|
|
50
51
|
"AUTORESEARCH_VERTEX_PROJECT",
|
|
51
52
|
"AUTORESEARCH_VERTEX_REGION",
|
|
@@ -165,19 +166,23 @@ def _setting(key: str, flag: str, environ: dict[str, str], from_file: dict[str,
|
|
|
165
166
|
return from_file.get(key, "")
|
|
166
167
|
|
|
167
168
|
|
|
168
|
-
def _home(environ: dict[str, str], cwd: Path) -> Path:
|
|
169
|
-
"""The
|
|
170
|
-
|
|
171
|
-
and
|
|
172
|
-
|
|
173
|
-
|
|
169
|
+
def _home(environ: dict[str, str], cwd: Path, *, local: bool, root: Path) -> Path:
|
|
170
|
+
"""The directory the loop runs from. On Slurm it must be a source
|
|
171
|
+
checkout (AUTORESEARCH_HOME, else the current directory): the chain
|
|
172
|
+
deploys from it and every job runs from a flight snapshot of its HEAD.
|
|
173
|
+
The local loop has no deploy step and runs the installed package, so it
|
|
174
|
+
uses a checkout when one is at hand and otherwise a `home` directory
|
|
175
|
+
under the state root, where flights and logs land."""
|
|
176
|
+
named = environ.get("OUTERLOOP_HOME") or environ.get("AUTORESEARCH_HOME")
|
|
177
|
+
home = Path(named).expanduser() if named else cwd
|
|
178
|
+
if (home / "scripts" / "tick_chain.sbatch").is_file():
|
|
179
|
+
return home
|
|
180
|
+
if local and not named:
|
|
181
|
+
return root / "home"
|
|
182
|
+
raise StartError(
|
|
183
|
+
f"{home} is not a source checkout (no scripts/tick_chain.sbatch); "
|
|
184
|
+
"run start from a clone of the kernel, or set OUTERLOOP_HOME to one"
|
|
174
185
|
)
|
|
175
|
-
if not (home / "scripts" / "tick_chain.sbatch").is_file():
|
|
176
|
-
raise StartError(
|
|
177
|
-
f"{home} is not an autoresearch checkout (no scripts/tick_chain.sbatch); "
|
|
178
|
-
"run start from the checkout the chain should deploy from, or set AUTORESEARCH_HOME"
|
|
179
|
-
)
|
|
180
|
-
return home
|
|
181
186
|
|
|
182
187
|
|
|
183
188
|
def plan_start(
|
|
@@ -207,13 +212,15 @@ def plan_start(
|
|
|
207
212
|
f"AUTORESEARCH_CADENCE_MIN must be a positive number of minutes, got {cadence!r}"
|
|
208
213
|
)
|
|
209
214
|
pat = _setting("AUTORESEARCH_PAT_FILE", "", environ, from_file)
|
|
210
|
-
#
|
|
211
|
-
#
|
|
212
|
-
|
|
215
|
+
# Slurm runs from a checkout; the local loop runs the installed package
|
|
216
|
+
# and needs only a directory (the launch lanes and GitHub servicing
|
|
217
|
+
# switch off without AUTORESEARCH_HOME, so one is always set)
|
|
218
|
+
local_root = Path(root_s).expanduser() if root_s else DEFAULT_LOCAL_ROOT
|
|
219
|
+
home = _home(environ, cwd, local=(mode == "local"), root=local_root)
|
|
213
220
|
if mode == "local":
|
|
214
221
|
return StartPlan(
|
|
215
222
|
mode="local",
|
|
216
|
-
root=
|
|
223
|
+
root=local_root,
|
|
217
224
|
home=home,
|
|
218
225
|
cadence_min=cadence,
|
|
219
226
|
pat_file=pat,
|
|
@@ -333,6 +340,7 @@ def start(args: argparse.Namespace) -> int:
|
|
|
333
340
|
env["AUTORESEARCH_COMPUTE"] = "local"
|
|
334
341
|
env["AUTORESEARCH_ROOT"] = str(plan.root)
|
|
335
342
|
env["AUTORESEARCH_HOME"] = str(plan.home)
|
|
343
|
+
plan.home.mkdir(parents=True, exist_ok=True) # <root>/home when there is no checkout
|
|
336
344
|
if plan.cadence_min:
|
|
337
345
|
env["AUTORESEARCH_CADENCE_MIN"] = plan.cadence_min
|
|
338
346
|
if plan.pat_file:
|