codex-orchestrator 0.1.30 → 0.1.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/README.md +152 -164
- package/dist/src/cli.js +68 -1
- package/dist/src/cli.js.map +1 -1
- package/dist/src/codex/command-adapter.d.ts +2 -1
- package/dist/src/codex/command-adapter.d.ts.map +1 -1
- package/dist/src/codex/command-adapter.js +13 -2
- package/dist/src/codex/command-adapter.js.map +1 -1
- package/dist/src/codex/mobile-device-guard.d.ts +7 -0
- package/dist/src/codex/mobile-device-guard.d.ts.map +1 -0
- package/dist/src/codex/mobile-device-guard.js +63 -0
- package/dist/src/codex/mobile-device-guard.js.map +1 -0
- package/dist/src/config/constants.d.ts +1 -1
- package/dist/src/config/constants.d.ts.map +1 -1
- package/dist/src/config/constants.js +1 -0
- package/dist/src/config/constants.js.map +1 -1
- package/dist/src/config/schema.d.ts +16 -2
- package/dist/src/config/schema.d.ts.map +1 -1
- package/dist/src/config/schema.js +42 -0
- package/dist/src/config/schema.js.map +1 -1
- package/dist/src/github/gh-pull-request-adapter.d.ts +1 -0
- package/dist/src/github/gh-pull-request-adapter.d.ts.map +1 -1
- package/dist/src/github/gh-pull-request-adapter.js +20 -0
- package/dist/src/github/gh-pull-request-adapter.js.map +1 -1
- package/dist/src/github/pull-requests.d.ts +2 -0
- package/dist/src/github/pull-requests.d.ts.map +1 -1
- package/dist/src/github/pull-requests.js +9 -0
- package/dist/src/github/pull-requests.js.map +1 -1
- package/dist/src/runner/acceptance-proof-runner.d.ts +54 -0
- package/dist/src/runner/acceptance-proof-runner.d.ts.map +1 -0
- package/dist/src/runner/acceptance-proof-runner.js +229 -0
- package/dist/src/runner/acceptance-proof-runner.js.map +1 -0
- package/dist/src/runner/acceptance-proof.d.ts +121 -0
- package/dist/src/runner/acceptance-proof.d.ts.map +1 -0
- package/dist/src/runner/acceptance-proof.js +405 -0
- package/dist/src/runner/acceptance-proof.js.map +1 -0
- package/dist/src/runner/android-visual-proof-command.d.ts +26 -0
- package/dist/src/runner/android-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/android-visual-proof-command.js +598 -0
- package/dist/src/runner/android-visual-proof-command.js.map +1 -0
- package/dist/src/runner/command-utils.d.ts.map +1 -1
- package/dist/src/runner/command-utils.js +55 -6
- package/dist/src/runner/command-utils.js.map +1 -1
- package/dist/src/runner/completion-report.d.ts +3 -1
- package/dist/src/runner/completion-report.d.ts.map +1 -1
- package/dist/src/runner/completion-report.js +2 -1
- package/dist/src/runner/completion-report.js.map +1 -1
- package/dist/src/runner/daemon-command.d.ts +1 -0
- package/dist/src/runner/daemon-command.d.ts.map +1 -1
- package/dist/src/runner/daemon-command.js +139 -14
- package/dist/src/runner/daemon-command.js.map +1 -1
- package/dist/src/runner/doctor-command.js +3 -3
- package/dist/src/runner/doctor-command.js.map +1 -1
- package/dist/src/runner/durable-run-summary.d.ts +3 -0
- package/dist/src/runner/durable-run-summary.d.ts.map +1 -1
- package/dist/src/runner/durable-run-summary.js +2 -0
- package/dist/src/runner/durable-run-summary.js.map +1 -1
- package/dist/src/runner/handoff-evidence.d.ts +5 -0
- package/dist/src/runner/handoff-evidence.d.ts.map +1 -1
- package/dist/src/runner/handoff-evidence.js +22 -0
- package/dist/src/runner/handoff-evidence.js.map +1 -1
- package/dist/src/runner/ios-visual-proof-command.d.ts +25 -0
- package/dist/src/runner/ios-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/ios-visual-proof-command.js +355 -0
- package/dist/src/runner/ios-visual-proof-command.js.map +1 -0
- package/dist/src/runner/lifecycle-events.d.ts +1 -1
- package/dist/src/runner/lifecycle-events.d.ts.map +1 -1
- package/dist/src/runner/local-execution-session.d.ts +20 -6
- package/dist/src/runner/local-execution-session.d.ts.map +1 -1
- package/dist/src/runner/local-execution-session.js +98 -11
- package/dist/src/runner/local-execution-session.js.map +1 -1
- package/dist/src/runner/local-state.d.ts +6 -0
- package/dist/src/runner/local-state.d.ts.map +1 -1
- package/dist/src/runner/local-state.js +14 -0
- package/dist/src/runner/local-state.js.map +1 -1
- package/dist/src/runner/mobile-device-lease.d.ts +11 -0
- package/dist/src/runner/mobile-device-lease.d.ts.map +1 -0
- package/dist/src/runner/mobile-device-lease.js +121 -0
- package/dist/src/runner/mobile-device-lease.js.map +1 -0
- package/dist/src/runner/mobile-visual-proof-command.d.ts +17 -0
- package/dist/src/runner/mobile-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/mobile-visual-proof-command.js +168 -0
- package/dist/src/runner/mobile-visual-proof-command.js.map +1 -0
- package/dist/src/runner/plan-auto-command.d.ts.map +1 -1
- package/dist/src/runner/plan-auto-command.js +58 -7
- package/dist/src/runner/plan-auto-command.js.map +1 -1
- package/dist/src/runner/prompt.js +2 -2
- package/dist/src/runner/prompt.js.map +1 -1
- package/dist/src/runner/recovery.d.ts +2 -1
- package/dist/src/runner/recovery.d.ts.map +1 -1
- package/dist/src/runner/recovery.js +24 -0
- package/dist/src/runner/recovery.js.map +1 -1
- package/dist/src/runner/review-gate-policy.d.ts +7 -0
- package/dist/src/runner/review-gate-policy.d.ts.map +1 -1
- package/dist/src/runner/review-gate-policy.js +88 -22
- package/dist/src/runner/review-gate-policy.js.map +1 -1
- package/dist/src/runner/review-gates.d.ts.map +1 -1
- package/dist/src/runner/review-gates.js +44 -11
- package/dist/src/runner/review-gates.js.map +1 -1
- package/dist/src/runner/rework-policy.d.ts +1 -0
- package/dist/src/runner/rework-policy.d.ts.map +1 -1
- package/dist/src/runner/rework-policy.js +7 -0
- package/dist/src/runner/rework-policy.js.map +1 -1
- package/dist/src/runner/scoped-auto-command.d.ts +46 -1
- package/dist/src/runner/scoped-auto-command.d.ts.map +1 -1
- package/dist/src/runner/scoped-auto-command.js +148 -48
- package/dist/src/runner/scoped-auto-command.js.map +1 -1
- package/dist/src/runner/scoped-recovery.d.ts +65 -0
- package/dist/src/runner/scoped-recovery.d.ts.map +1 -0
- package/dist/src/runner/scoped-recovery.js +484 -0
- package/dist/src/runner/scoped-recovery.js.map +1 -0
- package/dist/src/runner/status-command.d.ts.map +1 -1
- package/dist/src/runner/status-command.js +1 -0
- package/dist/src/runner/status-command.js.map +1 -1
- package/dist/src/runner/visual-proof-runner.d.ts +1 -0
- package/dist/src/runner/visual-proof-runner.d.ts.map +1 -1
- package/dist/src/runner/visual-proof-runner.js +144 -16
- package/dist/src/runner/visual-proof-runner.js.map +1 -1
- package/dist/src/setup/project-config.d.ts +4 -0
- package/dist/src/setup/project-config.d.ts.map +1 -1
- package/dist/src/setup/project-config.js +142 -9
- package/dist/src/setup/project-config.js.map +1 -1
- package/dist/src/setup/setup-command.d.ts.map +1 -1
- package/dist/src/setup/setup-command.js +2 -2
- package/dist/src/setup/setup-command.js.map +1 -1
- package/dist/src/setup/workflows.d.ts.map +1 -1
- package/dist/src/setup/workflows.js +5 -0
- package/dist/src/setup/workflows.js.map +1 -1
- package/docs/deep-dive.md +230 -31
- package/package.json +1 -1
- package/prompts/workflows/acceptance-proof.md +25 -0
- package/prompts/workflows/issue-breakdown.md +7 -0
- package/prompts/workflows/scoped-implementation.md +6 -2
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,71 @@ The format is based on Keep a Changelog, and this project follows SemVer.
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.35] - 2026-05-21
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- Added a runner-validated UI Evidence Contract for Acceptance Proof reports,
|
|
13
|
+
covering workflow, viewport, freshness, layout, copy, and source-input
|
|
14
|
+
evidence for screenshot and UI-dump artifacts.
|
|
15
|
+
- Added live smoke coverage for UI Evidence pass and blocking cases, including
|
|
16
|
+
missing UI Evidence and too-narrow desktop viewport proof.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
- Runner-owned visual proof no longer treats screenshot-only command success as
|
|
20
|
+
a pass path; proof commands must produce a valid machine-readable Acceptance
|
|
21
|
+
Proof report.
|
|
22
|
+
- Updated the legacy `visual-proof` live smoke scenario to emit the same
|
|
23
|
+
machine-readable UI Evidence report required by the runner.
|
|
24
|
+
|
|
25
|
+
## [0.1.34] - 2026-05-20
|
|
26
|
+
|
|
27
|
+
### Added
|
|
28
|
+
- Added an Adaptive Proof Agent Codex phase for scoped and issue-tree child
|
|
29
|
+
runs, with runner-provided proof report paths, artifact directories, changed
|
|
30
|
+
file context, and proof-owned repair policy.
|
|
31
|
+
- Added durable Acceptance Proof attempt evidence in lifecycle events, run
|
|
32
|
+
summaries, blocked comments, review reports, and issue-tree PR handoff.
|
|
33
|
+
- Added package-bundled Acceptance Proof workflow prompts and setup routing for
|
|
34
|
+
the new proof phase.
|
|
35
|
+
|
|
36
|
+
### Changed
|
|
37
|
+
- Parent `agent:plan-auto` child waves now block parent publication when a child
|
|
38
|
+
Acceptance Proof attempt fails, requests rework, or is blocked.
|
|
39
|
+
- Proof attempts now use isolated Codex homes and preserve proof artifacts while
|
|
40
|
+
keeping publication authority runner-owned.
|
|
41
|
+
|
|
42
|
+
## [0.1.33] - 2026-05-20
|
|
43
|
+
|
|
44
|
+
### Added
|
|
45
|
+
- Added canonical `reviewGates.acceptanceProof` policy with proof-owned path
|
|
46
|
+
classification and machine validation for high-confidence proof reports.
|
|
47
|
+
- Added live smoke scenarios for canonical Acceptance Proof pass, proof rework,
|
|
48
|
+
low-confidence blocking, and proof-phase product-diff blocking.
|
|
49
|
+
|
|
50
|
+
### Changed
|
|
51
|
+
- Kept `reviewGates.visualProof` as a compatibility adapter while routing
|
|
52
|
+
runner prompts and proof policy through Acceptance Proof.
|
|
53
|
+
- Proof-phase product-code changes now block publishability instead of being
|
|
54
|
+
silently committed as verification output.
|
|
55
|
+
|
|
56
|
+
## [0.1.32] - 2026-05-19
|
|
57
|
+
|
|
58
|
+
### Added
|
|
59
|
+
- Added package-owned `visual-proof mobile`, `visual-proof android`, and
|
|
60
|
+
`visual-proof ios` commands for reusable UI launch proof across installed
|
|
61
|
+
repositories.
|
|
62
|
+
- Mobile visual proof now supports Flutter Android, native Android, Flutter iOS,
|
|
63
|
+
and native iOS projects, with screenshots saved into runner proof artifacts.
|
|
64
|
+
|
|
65
|
+
### Changed
|
|
66
|
+
- Setup now defaults visual proof to
|
|
67
|
+
`codex-orchestrator visual-proof mobile --issue ${issueNumber}` instead of a
|
|
68
|
+
target-repo local proof script.
|
|
69
|
+
- Android proof resolves SDK tools from environment variables, `PATH`, and
|
|
70
|
+
default macOS, Linux, and Windows SDK locations.
|
|
71
|
+
- On macOS, mobile proof falls back to the iOS simulator when Android tooling or
|
|
72
|
+
devices are unavailable and the repo has an iOS target.
|
|
73
|
+
|
|
9
74
|
## [0.1.30] - 2026-05-18
|
|
10
75
|
|
|
11
76
|
### Added
|
package/README.md
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
# codex-orchestrator
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
## About
|
|
4
|
+
|
|
5
|
+
`codex-orchestrator` turns GitHub Issues and project work into isolated,
|
|
6
|
+
autonomous Codex implementation runs, allowing maintainers to manage work
|
|
7
|
+
instead of supervising coding agents.
|
|
4
8
|
|
|
5
9
|
Instead of starting a new Codex chat for every issue, you label the work you
|
|
6
10
|
want automated. The runner creates an isolated workspace, gives Codex the issue
|
|
@@ -41,23 +45,76 @@ control to humans before anything is merged.
|
|
|
41
45
|
- A repeatable way to send selected GitHub Issues to Codex.
|
|
42
46
|
- One-off autonomous runs for scoped implementation tasks.
|
|
43
47
|
- Parent planning for larger features, with child issues executed in safe waves.
|
|
44
|
-
- Project-owned rules for
|
|
45
|
-
|
|
46
|
-
-
|
|
47
|
-
and
|
|
48
|
-
-
|
|
49
|
-
|
|
48
|
+
- Project-owned rules for what Codex may run, how results are checked, and when
|
|
49
|
+
a human must step in.
|
|
50
|
+
- Adaptive Acceptance Proof for runner-owned verification of UI, API, worker,
|
|
51
|
+
CLI, browser, mobile, and live-smoke behavior before draft PR handoff.
|
|
52
|
+
- Logs, summaries, and proof artifacts when available.
|
|
53
|
+
- Recovery for interrupted runner handoff when Codex finished locally but the
|
|
54
|
+
draft PR was not created yet.
|
|
50
55
|
- Draft PR handoff by default. No auto-merge.
|
|
51
56
|
|
|
52
57
|
## How It Works
|
|
53
58
|
|
|
54
|
-
|
|
59
|
+
At a high level, GitHub Issues are the queue, labels authorize work, isolated
|
|
60
|
+
worktrees keep runs separate, and the runner owns validation and publication.
|
|
61
|
+
|
|
62
|
+
```mermaid
|
|
63
|
+
flowchart TD
|
|
64
|
+
A["Target repo"] --> B["codex-orchestrator setup"]
|
|
65
|
+
B --> C[".codex-orchestrator/config.json + prompts"]
|
|
66
|
+
C --> D["GitHub Issue gets agent:auto or agent:plan-auto"]
|
|
67
|
+
D --> E["status / daemon / run"]
|
|
68
|
+
E --> R{"Recover interrupted handoff?"}
|
|
69
|
+
R -- "yes" --> L
|
|
70
|
+
R -- "no" --> F{"Eligible?"}
|
|
71
|
+
F -- "no" --> G["Skipped with reason"]
|
|
72
|
+
F -- "yes" --> H["Runner claims issue: agent:running"]
|
|
73
|
+
H --> I["Create isolated branch + worktree"]
|
|
74
|
+
I --> J["Build Codex prompt from issue + repo policy"]
|
|
75
|
+
J --> K["Run Codex CLI"]
|
|
76
|
+
K --> AAP{"Adaptive Acceptance Proof required?"}
|
|
77
|
+
AAP -- "no" --> L["Runner validates the full changeset"]
|
|
78
|
+
AAP -- "yes" --> AP["Run proof phase and collect artifacts"]
|
|
79
|
+
AP --> APR{"Proof result"}
|
|
80
|
+
APR -- "passed" --> L
|
|
81
|
+
APR -- "needs rework" --> J
|
|
82
|
+
APR -- "blocked" --> N
|
|
83
|
+
L --> M{"Gates pass?"}
|
|
84
|
+
M -- "no" --> N["Mark blocked, preserve evidence"]
|
|
85
|
+
M -- "yes" --> O["Push branch"]
|
|
86
|
+
O --> P["Open draft PR"]
|
|
87
|
+
P --> Q["Move issue to agent:review + post report"]
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The important boundary is simple: Codex writes code, but the runner decides
|
|
91
|
+
whether that code can be handed to humans. The runner owns checks, acceptance
|
|
92
|
+
proof, labels, comments, branch pushes, and draft PR creation.
|
|
93
|
+
|
|
94
|
+
### Adaptive Acceptance Proof
|
|
95
|
+
|
|
96
|
+
Adaptive Acceptance Proof is the runner-owned verification phase for work that
|
|
97
|
+
needs observable product proof. After implementation, the runner can start a
|
|
98
|
+
separate proof phase that inspects the issue, changed files, and acceptance
|
|
99
|
+
criteria; runs focused browser, mobile, API, worker, CLI, or live-smoke checks;
|
|
100
|
+
and writes a machine-readable proof report with artifact links.
|
|
101
|
+
|
|
102
|
+
A result can reach draft PR handoff only when every required criterion maps to
|
|
103
|
+
high-confidence artifact evidence. If proof finds missing behavior, it returns a
|
|
104
|
+
concrete rework request and the runner loops back through implementation within
|
|
105
|
+
the configured iteration limit. If proof is malformed, low-confidence, lacks
|
|
106
|
+
artifacts, or changes product code during verification, the runner blocks
|
|
107
|
+
publication and preserves the evidence.
|
|
108
|
+
|
|
109
|
+
For UI proof, screenshots and UI dumps must also satisfy the UI Evidence
|
|
110
|
+
Contract: exact workflow, viewport coverage, current artifact freshness, layout
|
|
111
|
+
review, copy review, and source inputs. Screenshot-only proof cannot pass.
|
|
112
|
+
|
|
113
|
+
There are two main ways to run work.
|
|
55
114
|
|
|
56
115
|
### `agent:auto`
|
|
57
116
|
|
|
58
|
-
Use `agent:auto` for one clear standalone implementation issue.
|
|
59
|
-
for child issues created by `agent:plan-auto`; those are marked with
|
|
60
|
-
`agent:child` and are executed only by the parent issue-tree flow.
|
|
117
|
+
Use `agent:auto` for one clear standalone implementation issue.
|
|
61
118
|
|
|
62
119
|
The runner:
|
|
63
120
|
|
|
@@ -69,18 +126,26 @@ The runner:
|
|
|
69
126
|
6. Pushes the branch and opens a draft PR only after the gates pass.
|
|
70
127
|
7. Moves the issue to review and posts the run report.
|
|
71
128
|
|
|
129
|
+
When the daemon runs more than one `agent:auto` issue at once, it only batches
|
|
130
|
+
issues whose declared ownership does not overlap. Issues without ownership
|
|
131
|
+
metadata still run, but conservatively.
|
|
132
|
+
|
|
72
133
|
### `agent:plan-auto`
|
|
73
134
|
|
|
74
|
-
Use `agent:plan-auto` for work that
|
|
135
|
+
Use `agent:plan-auto` for larger work that should be planned before
|
|
136
|
+
implementation.
|
|
75
137
|
|
|
76
138
|
The runner asks Codex to plan the parent issue, break it into child issues,
|
|
77
|
-
|
|
78
|
-
integration draft
|
|
139
|
+
run safe children in dependency order, merge successful child branches into one
|
|
140
|
+
integration branch, validate that integration branch, and then open one draft
|
|
141
|
+
PR.
|
|
142
|
+
|
|
143
|
+
Child issues created by this flow use `agent:child`, not `agent:auto`. They are
|
|
144
|
+
owned by the parent run and are not picked up as standalone daemon work.
|
|
79
145
|
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
parent-owned child work with standalone scoped work.
|
|
146
|
+
If a runner stops after Codex finished locally but before draft PR handoff, the
|
|
147
|
+
runner can recover from its local state and completed report without rerunning
|
|
148
|
+
Codex. See [docs/deep-dive.md](docs/deep-dive.md) for the recovery rules.
|
|
84
149
|
|
|
85
150
|
## Basic Workflow
|
|
86
151
|
|
|
@@ -94,7 +159,13 @@ parent-owned child work with standalone scoped work.
|
|
|
94
159
|
|
|
95
160
|
The runner never auto-merges.
|
|
96
161
|
|
|
97
|
-
##
|
|
162
|
+
## Agent Memory
|
|
163
|
+
|
|
164
|
+
Repo-local Dreaming-lite memory lives in `docs/agents/memory/`. It is a small
|
|
165
|
+
curated lessons cache for repeated runner/debug/agent-workflow patterns, not a
|
|
166
|
+
replacement for `AGENTS.md`, ADRs, `docs/deep-dive.md`, or package prompts.
|
|
167
|
+
|
|
168
|
+
## Install
|
|
98
169
|
|
|
99
170
|
Requirements:
|
|
100
171
|
|
|
@@ -123,7 +194,7 @@ You can also run it with `npx`:
|
|
|
123
194
|
npx codex-orchestrator --help
|
|
124
195
|
```
|
|
125
196
|
|
|
126
|
-
##
|
|
197
|
+
## Set Up A Repository
|
|
127
198
|
|
|
128
199
|
Open the repository that should receive autonomous Codex work:
|
|
129
200
|
|
|
@@ -137,19 +208,19 @@ Run setup and create missing labels:
|
|
|
137
208
|
codex-orchestrator setup --prepare-labels
|
|
138
209
|
```
|
|
139
210
|
|
|
140
|
-
By default, setup reads the GitHub owner and repository name from `git remote
|
|
141
|
-
origin` and uses the current directory as the target repository. Use `--target`,
|
|
142
|
-
`--github-owner`, and `--github-repo` only when you need to override those
|
|
143
|
-
defaults.
|
|
144
|
-
|
|
145
211
|
Commit the generated `.codex-orchestrator/` directory to your repository. It is
|
|
146
212
|
the repository-owned policy for how autonomous work should run.
|
|
147
213
|
|
|
148
|
-
|
|
214
|
+
By default, setup reads the GitHub owner and repo from `git remote origin`. Use
|
|
215
|
+
`--target`, `--github-owner`, or `--github-repo` only when you need to override
|
|
216
|
+
that.
|
|
217
|
+
|
|
218
|
+
## Run Work
|
|
219
|
+
|
|
220
|
+
Check what the runner can see:
|
|
149
221
|
|
|
150
222
|
```sh
|
|
151
223
|
codex-orchestrator status --target .
|
|
152
|
-
codex-orchestrator status --target . --json
|
|
153
224
|
codex-orchestrator doctor --target .
|
|
154
225
|
```
|
|
155
226
|
|
|
@@ -165,6 +236,16 @@ Run the daemon:
|
|
|
165
236
|
codex-orchestrator daemon --target .
|
|
166
237
|
```
|
|
167
238
|
|
|
239
|
+
Run up to three independent scoped issues at once:
|
|
240
|
+
|
|
241
|
+
```sh
|
|
242
|
+
codex-orchestrator daemon --target . --concurrency 3
|
|
243
|
+
```
|
|
244
|
+
|
|
245
|
+
`status` and `doctor` are read-only. `run` executes one selected issue.
|
|
246
|
+
`daemon` polls for eligible work and starts safe runs according to the policy in
|
|
247
|
+
`.codex-orchestrator/config.json`.
|
|
248
|
+
|
|
168
249
|
## Agent-Assisted Setup
|
|
169
250
|
|
|
170
251
|
You do not need a long prompt. You can ask an agent:
|
|
@@ -193,126 +274,60 @@ working in the repository can find repository-local setup guidance.
|
|
|
193
274
|
Use `--dry-run` only when you want a preview without writing files or creating
|
|
194
275
|
labels.
|
|
195
276
|
|
|
196
|
-
##
|
|
277
|
+
## What The Runner Checks
|
|
197
278
|
|
|
198
|
-
|
|
279
|
+
Before a result becomes a draft PR, the runner checks the whole local result:
|
|
199
280
|
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
281
|
+
- committed changes;
|
|
282
|
+
- staged changes;
|
|
283
|
+
- unstaged changes;
|
|
284
|
+
- untracked files;
|
|
285
|
+
- the completion report Codex was required to write;
|
|
286
|
+
- configured commands such as tests or type checks;
|
|
287
|
+
- review gates such as TDD evidence, changed tests, cleanup review, code review,
|
|
288
|
+
or acceptance proof when enabled;
|
|
289
|
+
- blocked paths and unsafe actions.
|
|
203
290
|
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
and the prompts used for planning and implementation.
|
|
208
|
-
|
|
209
|
-
The package ships bundled workflow prompts, so a repository does not need local
|
|
210
|
-
Codex `SKILL.md` files installed on the user's machine. Setup copies those
|
|
211
|
-
prompts into `.codex-orchestrator/prompts/workflows/`, and the runner reads the
|
|
212
|
-
copied prompt files during `agent:auto` and `agent:plan-auto` runs. Workflow
|
|
213
|
-
`skillName` values in config are descriptive metadata for the workflow role;
|
|
214
|
-
they are not a runtime dependency on the user's local Codex skill directory.
|
|
215
|
-
Setup also writes `.codex-orchestrator/prompts/manifest.json`, which lets later
|
|
216
|
-
setup runs tell apart untouched package prompts from prompts edited by the
|
|
217
|
-
project. By default, setup refreshes untouched prompts and reports conflicts for
|
|
218
|
-
locally edited prompts.
|
|
219
|
-
|
|
220
|
-
Configured checks run before publication. By default, missing
|
|
221
|
-
`npm run <script>` checks are reported as skipped warnings, not failures. You can
|
|
222
|
-
change that with `checksPolicy.missingNpmScript`.
|
|
223
|
-
|
|
224
|
-
For repos with existing lint debt, `checksPolicy.lintBaseline.mode` can be set
|
|
225
|
-
to `touched-only`. That lets a repo-wide lint failure be downgraded when a
|
|
226
|
-
separate touched-files lint command passes.
|
|
227
|
-
|
|
228
|
-
The default quality gate is conservative for runtime code changes. It can
|
|
229
|
-
require TDD evidence, changed tests, code review, cleanup review for larger
|
|
230
|
-
changes, and visual proof for UI work.
|
|
231
|
-
|
|
232
|
-
## Diagnostics
|
|
233
|
-
|
|
234
|
-
`doctor` is a read-only readiness check for operators. It validates the target
|
|
235
|
-
config, GitHub label visibility, git/base branch access, runner state paths,
|
|
236
|
-
configured checks, the Codex command, phase profiles, and visual proof settings.
|
|
237
|
-
It never launches Codex, creates worktrees, edits labels, or changes issues.
|
|
291
|
+
If the result passes, the runner pushes the branch and opens a draft PR. If it
|
|
292
|
+
does not pass, the runner marks the issue blocked, keeps the useful local
|
|
293
|
+
evidence, and explains what needs attention.
|
|
238
294
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
295
|
+
Acceptance proof is runner-owned. Codex can change product behavior, but the
|
|
296
|
+
runner runs proof afterwards and attaches screenshots, UI dumps, logs, smoke
|
|
297
|
+
outputs, or other artifacts to the PR and issue report. The proof phase must
|
|
298
|
+
produce a structured report that maps each required criterion to high-confidence
|
|
299
|
+
evidence. UI artifacts must include UI Evidence Contract mapping, and legacy
|
|
300
|
+
visual proof config only supplies migration inputs for report-producing proof.
|
|
243
301
|
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
not raw Codex transcripts, secrets, prompt text, or full issue comments.
|
|
248
|
-
|
|
249
|
-
Codex command profiles can be set per runner phase under `codex.profiles`.
|
|
250
|
-
Supported phases are `plan-parent`, `scoped-issue`, `tree-child`,
|
|
251
|
-
`fresh-context-review`, `visual-proof`, and `quality-review`. Missing profile
|
|
252
|
-
fields fall back to the global `codex.command`, `codex.args`, `timeoutMs`, and
|
|
253
|
-
`idleTimeoutMs`, so existing configs keep working.
|
|
254
|
-
|
|
255
|
-
Each Codex session writes a bounded context snapshot before invocation and links
|
|
256
|
-
it from lifecycle events under the runner state directory. Snapshots record the
|
|
257
|
-
issue identity, runner decision, selected profile, workspace paths, and
|
|
258
|
-
publication boundaries so a maintainer can reproduce why a session started
|
|
259
|
-
without reading raw logs.
|
|
260
|
-
|
|
261
|
-
## Visual Proof
|
|
262
|
-
|
|
263
|
-
For browser UI work, configure a runner-owned proof command, usually a
|
|
264
|
-
Playwright script:
|
|
265
|
-
|
|
266
|
-
```json
|
|
267
|
-
{
|
|
268
|
-
"reviewGates": {
|
|
269
|
-
"visualProof": {
|
|
270
|
-
"runnerValidationCommand": "npm run visual-proof -- --issue ${issueNumber}",
|
|
271
|
-
"runnerTimeoutMs": 900000,
|
|
272
|
-
"envPassthrough": [
|
|
273
|
-
"CODEX_ORCHESTRATOR_LOGIN_EMAIL",
|
|
274
|
-
"CODEX_ORCHESTRATOR_LOGIN_PASSWORD"
|
|
275
|
-
]
|
|
276
|
-
}
|
|
277
|
-
}
|
|
278
|
-
}
|
|
279
|
-
```
|
|
302
|
+
## Repository Policy
|
|
303
|
+
|
|
304
|
+
Every installed repository owns its automation policy in:
|
|
280
305
|
|
|
281
|
-
|
|
282
|
-
|
|
283
|
-
|
|
284
|
-
worktree path, and changed files.
|
|
306
|
+
```sh
|
|
307
|
+
.codex-orchestrator/config.json
|
|
308
|
+
```
|
|
285
309
|
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
310
|
+
That config controls labels, branch names, checks, review gates, blocked paths,
|
|
311
|
+
prompt files, child concurrency, and PR titles. The package provides defaults;
|
|
312
|
+
the target repository decides how strict they should be.
|
|
289
313
|
|
|
290
|
-
For
|
|
291
|
-
|
|
292
|
-
no usable device is reported as a warning with the concrete reason, not as an
|
|
293
|
-
automatic release blocker. Native Android proof uses the project Gradle wrapper
|
|
294
|
-
with a writable Gradle cache; Flutter-specific SDK cache recovery is used only
|
|
295
|
-
for Flutter projects and only through a preconfigured writable SDK path in
|
|
296
|
-
`CODEX_ORCHESTRATOR_FLUTTER_ROOT`. Native iOS proof uses Xcode simulator/device
|
|
297
|
-
tooling with a writable DerivedData path.
|
|
314
|
+
For the full config surface and technical behavior, see
|
|
315
|
+
[docs/deep-dive.md](docs/deep-dive.md).
|
|
298
316
|
|
|
299
317
|
## Safety Model
|
|
300
318
|
|
|
301
319
|
The package is PR-first and human-reviewed. The important guardrails are:
|
|
302
320
|
|
|
303
|
-
- no automatic merge
|
|
321
|
+
- no automatic merge;
|
|
322
|
+
- draft PRs only;
|
|
304
323
|
- Codex may change files, but the runner owns remote publication and GitHub
|
|
305
|
-
state;
|
|
324
|
+
state changes;
|
|
306
325
|
- only explicitly authorized issues run;
|
|
307
326
|
- child issues are never inferred from ordinary links or references;
|
|
308
|
-
- committed and uncommitted changes are checked
|
|
309
|
-
-
|
|
310
|
-
|
|
311
|
-
|
|
312
|
-
- bounded rework stops at the configured limit;
|
|
313
|
-
- Policy Suggestions are recommendations only;
|
|
314
|
-
- underspecified work can be blocked for maintainer clarification instead of
|
|
315
|
-
letting Codex invent product decisions.
|
|
327
|
+
- committed and uncommitted changes are checked;
|
|
328
|
+
- missing or malformed completion reports block publication;
|
|
329
|
+
- secret files, destructive data/cache actions, and production deploy or release
|
|
330
|
+
actions are blocked by default.
|
|
316
331
|
|
|
317
332
|
## Labels
|
|
318
333
|
|
|
@@ -340,42 +355,15 @@ codex-orchestrator setup [--target <path>] [--github-owner <owner>] \
|
|
|
340
355
|
[--github-repo <repo>] [--dry-run] [--prepare-labels]
|
|
341
356
|
codex-orchestrator status --target <path> [--dry-run] [--json]
|
|
342
357
|
codex-orchestrator run --target <path> --issue <number>
|
|
358
|
+
codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
|
|
359
|
+
codex-orchestrator visual-proof android --issue <number> [--target <path>]
|
|
360
|
+
codex-orchestrator visual-proof ios --issue <number> [--target <path>]
|
|
343
361
|
codex-orchestrator daemon --target <path> [--once] \
|
|
344
|
-
[--interval-seconds <seconds>] [--max-runs <count>]
|
|
362
|
+
[--interval-seconds <seconds>] [--max-runs <count>] \
|
|
363
|
+
[--concurrency <count>]
|
|
345
364
|
```
|
|
346
365
|
|
|
347
|
-
`
|
|
348
|
-
`.codex-orchestrator/`. Useful flags:
|
|
349
|
-
|
|
350
|
-
- `--dry-run` - show the setup plan without writing files or creating labels;
|
|
351
|
-
- `--prepare-labels` - create missing GitHub labels;
|
|
352
|
-
- `--target <path>` - override the target directory, which defaults to the current directory;
|
|
353
|
-
- `--github-owner <owner>` - override the GitHub owner inferred from `origin`;
|
|
354
|
-
- `--github-repo <repo>` - override the GitHub repo inferred from `origin`;
|
|
355
|
-
- `--sync-prompts <auto|keep|replace|merge>` - choose how package-bundled
|
|
356
|
-
prompt updates are applied. `auto` refreshes untouched prompts and reports
|
|
357
|
-
local-edit conflicts, `keep` preserves existing prompts, `replace` overwrites
|
|
358
|
-
with bundled prompts, and `merge` appends bundled updates to locally edited
|
|
359
|
-
prompts;
|
|
360
|
-
- `--replace-package-skills` - refresh package-bundled prompt files. The flag
|
|
361
|
-
name is kept for compatibility and behaves like `--sync-prompts=replace`; it
|
|
362
|
-
does not install or require local Codex skills.
|
|
363
|
-
|
|
364
|
-
Setup does not launch Codex, commit changes, or open pull requests.
|
|
365
|
-
When the target repository already has a `package.json`, setup also adds
|
|
366
|
-
`orchestrator:*` npm scripts. Daemon scripts run `doctor` first, then start the
|
|
367
|
-
daemon only if the readiness check passes.
|
|
368
|
-
|
|
369
|
-
`status` is read-only. It shows eligible issues, skipped issues with reasons,
|
|
370
|
-
and local recovery state.
|
|
371
|
-
|
|
372
|
-
`run` executes one selected issue when labels and state allow it. `agent:auto`
|
|
373
|
-
opens one scoped draft PR. `agent:plan-auto` runs parent planning, child waves,
|
|
374
|
-
final validation, and one integration draft PR.
|
|
375
|
-
|
|
376
|
-
`daemon` polls for eligible work and runs one issue at a time. It also cleans up
|
|
377
|
-
runner-owned worktrees after their PRs are merged, while preserving dirty,
|
|
378
|
-
blocked, active, or unpublished worktrees for inspection.
|
|
366
|
+
Use `codex-orchestrator <command> --help` for command-specific flags.
|
|
379
367
|
|
|
380
368
|
## Current Scope
|
|
381
369
|
|
package/dist/src/cli.js
CHANGED
|
@@ -8,7 +8,11 @@ import { runDaemonCommand } from './runner/daemon-command.js';
|
|
|
8
8
|
import { runDoctorCommand } from './runner/doctor-command.js';
|
|
9
9
|
import { runPlanAutoCommand } from './runner/plan-auto-command.js';
|
|
10
10
|
import { runScopedAutoCommand } from './runner/scoped-auto-command.js';
|
|
11
|
+
import { recoverScopedRun } from './runner/scoped-recovery.js';
|
|
11
12
|
import { runStatusCommand } from './runner/status-command.js';
|
|
13
|
+
import { parseAndroidVisualProofArgs, runAndroidVisualProofCommand } from './runner/android-visual-proof-command.js';
|
|
14
|
+
import { parseIosVisualProofArgs, runIosVisualProofCommand } from './runner/ios-visual-proof-command.js';
|
|
15
|
+
import { parseMobileVisualProofArgs, runMobileVisualProofCommand } from './runner/mobile-visual-proof-command.js';
|
|
12
16
|
import { runSetupCommand } from './setup/setup-command.js';
|
|
13
17
|
import { promptSyncModes } from './setup/prompt-sync.js';
|
|
14
18
|
const helpText = `codex-orchestrator
|
|
@@ -21,7 +25,10 @@ Usage:
|
|
|
21
25
|
codex-orchestrator setup [--target <path>] [--github-owner <owner>] [--github-repo <repo>] [--dry-run] [--prepare-labels] [--sync-prompts <mode>]
|
|
22
26
|
codex-orchestrator status --target <path> [--dry-run] [--json]
|
|
23
27
|
codex-orchestrator run --target <path> --issue <number>
|
|
24
|
-
codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>]
|
|
28
|
+
codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>] [--concurrency <number>]
|
|
29
|
+
codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
|
|
30
|
+
codex-orchestrator visual-proof android --issue <number> [--target <path>]
|
|
31
|
+
codex-orchestrator visual-proof ios --issue <number> [--target <path>]
|
|
25
32
|
|
|
26
33
|
Commands:
|
|
27
34
|
health Run a no-op local health check.
|
|
@@ -30,6 +37,7 @@ Commands:
|
|
|
30
37
|
status Show eligible/skipped issue work and local recovery state.
|
|
31
38
|
run Execute one authorized issue: scoped agent:auto or full agent:plan-auto issue tree.
|
|
32
39
|
daemon Poll GitHub Issues and execute eligible autonomous work until stopped.
|
|
40
|
+
visual-proof Run package-owned proof commands used by review gates.
|
|
33
41
|
|
|
34
42
|
Options:
|
|
35
43
|
--help, -h Show this help.
|
|
@@ -151,6 +159,7 @@ async function main(args) {
|
|
|
151
159
|
intervalMs: parsed.value.intervalSeconds * 1000,
|
|
152
160
|
once: parsed.value.once,
|
|
153
161
|
maxRuns: parsed.value.maxRuns,
|
|
162
|
+
concurrency: parsed.value.concurrency,
|
|
154
163
|
onEvent: (line) => {
|
|
155
164
|
process.stdout.write(`${line}\n`);
|
|
156
165
|
},
|
|
@@ -163,6 +172,48 @@ async function main(args) {
|
|
|
163
172
|
return 1;
|
|
164
173
|
}
|
|
165
174
|
}
|
|
175
|
+
if (command === 'visual-proof') {
|
|
176
|
+
const [kind, ...rest] = args.slice(1);
|
|
177
|
+
if (kind !== 'mobile' && kind !== 'android' && kind !== 'ios') {
|
|
178
|
+
process.stderr.write('visual-proof requires a supported kind: mobile, android, or ios\nRun codex-orchestrator --help for usage.\n');
|
|
179
|
+
return 2;
|
|
180
|
+
}
|
|
181
|
+
try {
|
|
182
|
+
if (kind === 'mobile') {
|
|
183
|
+
const parsed = parseMobileVisualProofArgs(rest);
|
|
184
|
+
if (!parsed.ok) {
|
|
185
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
186
|
+
return 2;
|
|
187
|
+
}
|
|
188
|
+
await runMobileVisualProofCommand(parsed.value);
|
|
189
|
+
process.stdout.write(`mobile visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
190
|
+
}
|
|
191
|
+
else if (kind === 'ios') {
|
|
192
|
+
const parsed = parseIosVisualProofArgs(rest);
|
|
193
|
+
if (!parsed.ok) {
|
|
194
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
195
|
+
return 2;
|
|
196
|
+
}
|
|
197
|
+
await runIosVisualProofCommand(parsed.value);
|
|
198
|
+
process.stdout.write(`ios visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
199
|
+
}
|
|
200
|
+
else {
|
|
201
|
+
const parsed = parseAndroidVisualProofArgs(rest);
|
|
202
|
+
if (!parsed.ok) {
|
|
203
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
204
|
+
return 2;
|
|
205
|
+
}
|
|
206
|
+
await runAndroidVisualProofCommand(parsed.value);
|
|
207
|
+
process.stdout.write(`android visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
208
|
+
}
|
|
209
|
+
return 0;
|
|
210
|
+
}
|
|
211
|
+
catch (error) {
|
|
212
|
+
const message = error instanceof Error ? error.message : 'visual proof failed';
|
|
213
|
+
process.stderr.write(`${message}\n`);
|
|
214
|
+
return 1;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
166
217
|
process.stderr.write(`Unknown command: ${command}\nRun codex-orchestrator --help for usage.\n`);
|
|
167
218
|
return 1;
|
|
168
219
|
}
|
|
@@ -176,6 +227,15 @@ async function runIssueCommand(targetRootInput, issueNumber) {
|
|
|
176
227
|
}
|
|
177
228
|
const decision = discoverIssueWork([issue], config)[0];
|
|
178
229
|
if (!decision || decision.kind !== 'eligible') {
|
|
230
|
+
const recovered = await recoverScopedRun({
|
|
231
|
+
targetRoot,
|
|
232
|
+
issueNumber,
|
|
233
|
+
invocation: 'targeted',
|
|
234
|
+
issueAdapter,
|
|
235
|
+
});
|
|
236
|
+
if (recovered.status !== 'not-recoverable') {
|
|
237
|
+
return { reportComment: recovered.reportComment };
|
|
238
|
+
}
|
|
179
239
|
const reason = decision?.kind === 'skipped' ? decision.reason : 'not eligible';
|
|
180
240
|
throw new Error(`Issue #${issueNumber} is not eligible for autonomous work: ${reason}`);
|
|
181
241
|
}
|
|
@@ -307,6 +367,13 @@ function parseDaemonArgs(args) {
|
|
|
307
367
|
parsed.maxRuns = Number(next);
|
|
308
368
|
index += 1;
|
|
309
369
|
break;
|
|
370
|
+
case '--concurrency':
|
|
371
|
+
if (!next || next.startsWith('--') || !Number.isInteger(Number(next)) || Number(next) < 1 || Number(next) > 3) {
|
|
372
|
+
return { ok: false, error: 'daemon requires --concurrency <integer between 1 and 3>' };
|
|
373
|
+
}
|
|
374
|
+
parsed.concurrency = Number(next);
|
|
375
|
+
index += 1;
|
|
376
|
+
break;
|
|
310
377
|
default:
|
|
311
378
|
return { ok: false, error: `Unknown daemon option: ${arg ?? ''}` };
|
|
312
379
|
}
|