codex-orchestrator 0.1.31 → 0.1.35
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/README.md +145 -174
- package/dist/src/cli.js +59 -0
- package/dist/src/cli.js.map +1 -1
- package/dist/src/codex/command-adapter.d.ts +2 -1
- package/dist/src/codex/command-adapter.d.ts.map +1 -1
- package/dist/src/codex/command-adapter.js +13 -2
- package/dist/src/codex/command-adapter.js.map +1 -1
- package/dist/src/codex/mobile-device-guard.d.ts +7 -0
- package/dist/src/codex/mobile-device-guard.d.ts.map +1 -0
- package/dist/src/codex/mobile-device-guard.js +63 -0
- package/dist/src/codex/mobile-device-guard.js.map +1 -0
- package/dist/src/config/constants.d.ts +1 -1
- package/dist/src/config/constants.d.ts.map +1 -1
- package/dist/src/config/constants.js +1 -0
- package/dist/src/config/constants.js.map +1 -1
- package/dist/src/config/schema.d.ts +15 -2
- package/dist/src/config/schema.d.ts.map +1 -1
- package/dist/src/config/schema.js +30 -0
- package/dist/src/config/schema.js.map +1 -1
- package/dist/src/github/gh-pull-request-adapter.d.ts +1 -0
- package/dist/src/github/gh-pull-request-adapter.d.ts.map +1 -1
- package/dist/src/github/gh-pull-request-adapter.js +20 -0
- package/dist/src/github/gh-pull-request-adapter.js.map +1 -1
- package/dist/src/github/pull-requests.d.ts +2 -0
- package/dist/src/github/pull-requests.d.ts.map +1 -1
- package/dist/src/github/pull-requests.js +9 -0
- package/dist/src/github/pull-requests.js.map +1 -1
- package/dist/src/runner/acceptance-proof-runner.d.ts +54 -0
- package/dist/src/runner/acceptance-proof-runner.d.ts.map +1 -0
- package/dist/src/runner/acceptance-proof-runner.js +229 -0
- package/dist/src/runner/acceptance-proof-runner.js.map +1 -0
- package/dist/src/runner/acceptance-proof.d.ts +121 -0
- package/dist/src/runner/acceptance-proof.d.ts.map +1 -0
- package/dist/src/runner/acceptance-proof.js +405 -0
- package/dist/src/runner/acceptance-proof.js.map +1 -0
- package/dist/src/runner/android-visual-proof-command.d.ts +26 -0
- package/dist/src/runner/android-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/android-visual-proof-command.js +598 -0
- package/dist/src/runner/android-visual-proof-command.js.map +1 -0
- package/dist/src/runner/command-utils.d.ts.map +1 -1
- package/dist/src/runner/command-utils.js +55 -6
- package/dist/src/runner/command-utils.js.map +1 -1
- package/dist/src/runner/completion-report.d.ts +3 -1
- package/dist/src/runner/completion-report.d.ts.map +1 -1
- package/dist/src/runner/completion-report.js +2 -1
- package/dist/src/runner/completion-report.js.map +1 -1
- package/dist/src/runner/daemon-command.d.ts.map +1 -1
- package/dist/src/runner/daemon-command.js +20 -0
- package/dist/src/runner/daemon-command.js.map +1 -1
- package/dist/src/runner/doctor-command.js +3 -3
- package/dist/src/runner/doctor-command.js.map +1 -1
- package/dist/src/runner/durable-run-summary.d.ts +3 -0
- package/dist/src/runner/durable-run-summary.d.ts.map +1 -1
- package/dist/src/runner/durable-run-summary.js +2 -0
- package/dist/src/runner/durable-run-summary.js.map +1 -1
- package/dist/src/runner/handoff-evidence.d.ts +5 -0
- package/dist/src/runner/handoff-evidence.d.ts.map +1 -1
- package/dist/src/runner/handoff-evidence.js +22 -0
- package/dist/src/runner/handoff-evidence.js.map +1 -1
- package/dist/src/runner/ios-visual-proof-command.d.ts +25 -0
- package/dist/src/runner/ios-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/ios-visual-proof-command.js +355 -0
- package/dist/src/runner/ios-visual-proof-command.js.map +1 -0
- package/dist/src/runner/lifecycle-events.d.ts +1 -1
- package/dist/src/runner/lifecycle-events.d.ts.map +1 -1
- package/dist/src/runner/local-execution-session.d.ts +19 -6
- package/dist/src/runner/local-execution-session.d.ts.map +1 -1
- package/dist/src/runner/local-execution-session.js +93 -11
- package/dist/src/runner/local-execution-session.js.map +1 -1
- package/dist/src/runner/local-state.d.ts +6 -0
- package/dist/src/runner/local-state.d.ts.map +1 -1
- package/dist/src/runner/local-state.js +14 -0
- package/dist/src/runner/local-state.js.map +1 -1
- package/dist/src/runner/mobile-device-lease.d.ts +11 -0
- package/dist/src/runner/mobile-device-lease.d.ts.map +1 -0
- package/dist/src/runner/mobile-device-lease.js +121 -0
- package/dist/src/runner/mobile-device-lease.js.map +1 -0
- package/dist/src/runner/mobile-visual-proof-command.d.ts +17 -0
- package/dist/src/runner/mobile-visual-proof-command.d.ts.map +1 -0
- package/dist/src/runner/mobile-visual-proof-command.js +168 -0
- package/dist/src/runner/mobile-visual-proof-command.js.map +1 -0
- package/dist/src/runner/plan-auto-command.d.ts.map +1 -1
- package/dist/src/runner/plan-auto-command.js +57 -6
- package/dist/src/runner/plan-auto-command.js.map +1 -1
- package/dist/src/runner/prompt.js +2 -2
- package/dist/src/runner/prompt.js.map +1 -1
- package/dist/src/runner/recovery.d.ts +2 -1
- package/dist/src/runner/recovery.d.ts.map +1 -1
- package/dist/src/runner/recovery.js +24 -0
- package/dist/src/runner/recovery.js.map +1 -1
- package/dist/src/runner/review-gate-policy.d.ts +7 -0
- package/dist/src/runner/review-gate-policy.d.ts.map +1 -1
- package/dist/src/runner/review-gate-policy.js +81 -20
- package/dist/src/runner/review-gate-policy.js.map +1 -1
- package/dist/src/runner/review-gates.d.ts.map +1 -1
- package/dist/src/runner/review-gates.js +44 -11
- package/dist/src/runner/review-gates.js.map +1 -1
- package/dist/src/runner/rework-policy.d.ts +1 -0
- package/dist/src/runner/rework-policy.d.ts.map +1 -1
- package/dist/src/runner/rework-policy.js +7 -0
- package/dist/src/runner/rework-policy.js.map +1 -1
- package/dist/src/runner/scoped-auto-command.d.ts +46 -1
- package/dist/src/runner/scoped-auto-command.d.ts.map +1 -1
- package/dist/src/runner/scoped-auto-command.js +147 -47
- package/dist/src/runner/scoped-auto-command.js.map +1 -1
- package/dist/src/runner/scoped-recovery.d.ts +65 -0
- package/dist/src/runner/scoped-recovery.d.ts.map +1 -0
- package/dist/src/runner/scoped-recovery.js +484 -0
- package/dist/src/runner/scoped-recovery.js.map +1 -0
- package/dist/src/runner/status-command.d.ts.map +1 -1
- package/dist/src/runner/status-command.js +1 -0
- package/dist/src/runner/status-command.js.map +1 -1
- package/dist/src/runner/visual-proof-runner.d.ts +1 -0
- package/dist/src/runner/visual-proof-runner.d.ts.map +1 -1
- package/dist/src/runner/visual-proof-runner.js +144 -16
- package/dist/src/runner/visual-proof-runner.js.map +1 -1
- package/dist/src/setup/project-config.d.ts +4 -0
- package/dist/src/setup/project-config.d.ts.map +1 -1
- package/dist/src/setup/project-config.js +141 -9
- package/dist/src/setup/project-config.js.map +1 -1
- package/dist/src/setup/setup-command.d.ts.map +1 -1
- package/dist/src/setup/setup-command.js +2 -2
- package/dist/src/setup/setup-command.js.map +1 -1
- package/dist/src/setup/workflows.d.ts.map +1 -1
- package/dist/src/setup/workflows.js +5 -0
- package/dist/src/setup/workflows.js.map +1 -1
- package/docs/deep-dive.md +217 -28
- package/package.json +1 -1
- package/prompts/workflows/acceptance-proof.md +25 -0
- package/prompts/workflows/scoped-implementation.md +6 -2
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,71 @@ The format is based on Keep a Changelog, and this project follows SemVer.
|
|
|
6
6
|
|
|
7
7
|
## [Unreleased]
|
|
8
8
|
|
|
9
|
+
## [0.1.35] - 2026-05-21
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- Added a runner-validated UI Evidence Contract for Acceptance Proof reports,
|
|
13
|
+
covering workflow, viewport, freshness, layout, copy, and source-input
|
|
14
|
+
evidence for screenshot and UI-dump artifacts.
|
|
15
|
+
- Added live smoke coverage for UI Evidence pass and blocking cases, including
|
|
16
|
+
missing UI Evidence and too-narrow desktop viewport proof.
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
- Runner-owned visual proof no longer treats screenshot-only command success as
|
|
20
|
+
a pass path; proof commands must produce a valid machine-readable Acceptance
|
|
21
|
+
Proof report.
|
|
22
|
+
- Updated the legacy `visual-proof` live smoke scenario to emit the same
|
|
23
|
+
machine-readable UI Evidence report required by the runner.
|
|
24
|
+
|
|
25
|
+
## [0.1.34] - 2026-05-20
|
|
26
|
+
|
|
27
|
+
### Added
|
|
28
|
+
- Added an Adaptive Proof Agent Codex phase for scoped and issue-tree child
|
|
29
|
+
runs, with runner-provided proof report paths, artifact directories, changed
|
|
30
|
+
file context, and proof-owned repair policy.
|
|
31
|
+
- Added durable Acceptance Proof attempt evidence in lifecycle events, run
|
|
32
|
+
summaries, blocked comments, review reports, and issue-tree PR handoff.
|
|
33
|
+
- Added package-bundled Acceptance Proof workflow prompts and setup routing for
|
|
34
|
+
the new proof phase.
|
|
35
|
+
|
|
36
|
+
### Changed
|
|
37
|
+
- Parent `agent:plan-auto` child waves now block parent publication when a child
|
|
38
|
+
Acceptance Proof attempt fails, requests rework, or is blocked.
|
|
39
|
+
- Proof attempts now use isolated Codex homes and preserve proof artifacts while
|
|
40
|
+
keeping publication authority runner-owned.
|
|
41
|
+
|
|
42
|
+
## [0.1.33] - 2026-05-20
|
|
43
|
+
|
|
44
|
+
### Added
|
|
45
|
+
- Added canonical `reviewGates.acceptanceProof` policy with proof-owned path
|
|
46
|
+
classification and machine validation for high-confidence proof reports.
|
|
47
|
+
- Added live smoke scenarios for canonical Acceptance Proof pass, proof rework,
|
|
48
|
+
low-confidence blocking, and proof-phase product-diff blocking.
|
|
49
|
+
|
|
50
|
+
### Changed
|
|
51
|
+
- Kept `reviewGates.visualProof` as a compatibility adapter while routing
|
|
52
|
+
runner prompts and proof policy through Acceptance Proof.
|
|
53
|
+
- Proof-phase product-code changes now block publishability instead of being
|
|
54
|
+
silently committed as verification output.
|
|
55
|
+
|
|
56
|
+
## [0.1.32] - 2026-05-19
|
|
57
|
+
|
|
58
|
+
### Added
|
|
59
|
+
- Added package-owned `visual-proof mobile`, `visual-proof android`, and
|
|
60
|
+
`visual-proof ios` commands for reusable UI launch proof across installed
|
|
61
|
+
repositories.
|
|
62
|
+
- Mobile visual proof now supports Flutter Android, native Android, Flutter iOS,
|
|
63
|
+
and native iOS projects, with screenshots saved into runner proof artifacts.
|
|
64
|
+
|
|
65
|
+
### Changed
|
|
66
|
+
- Setup now defaults visual proof to
|
|
67
|
+
`codex-orchestrator visual-proof mobile --issue ${issueNumber}` instead of a
|
|
68
|
+
target-repo local proof script.
|
|
69
|
+
- Android proof resolves SDK tools from environment variables, `PATH`, and
|
|
70
|
+
default macOS, Linux, and Windows SDK locations.
|
|
71
|
+
- On macOS, mobile proof falls back to the iOS simulator when Android tooling or
|
|
72
|
+
devices are unavailable and the repo has an iOS target.
|
|
73
|
+
|
|
9
74
|
## [0.1.30] - 2026-05-18
|
|
10
75
|
|
|
11
76
|
### Added
|
package/README.md
CHANGED
|
@@ -1,6 +1,10 @@
|
|
|
1
1
|
# codex-orchestrator
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
## About
|
|
4
|
+
|
|
5
|
+
`codex-orchestrator` turns GitHub Issues and project work into isolated,
|
|
6
|
+
autonomous Codex implementation runs, allowing maintainers to manage work
|
|
7
|
+
instead of supervising coding agents.
|
|
4
8
|
|
|
5
9
|
Instead of starting a new Codex chat for every issue, you label the work you
|
|
6
10
|
want automated. The runner creates an isolated workspace, gives Codex the issue
|
|
@@ -41,29 +45,76 @@ control to humans before anything is merged.
|
|
|
41
45
|
- A repeatable way to send selected GitHub Issues to Codex.
|
|
42
46
|
- One-off autonomous runs for scoped implementation tasks.
|
|
43
47
|
- Parent planning for larger features, with child issues executed in safe waves.
|
|
44
|
-
- Project-owned rules for
|
|
45
|
-
|
|
46
|
-
-
|
|
47
|
-
and
|
|
48
|
-
-
|
|
49
|
-
|
|
48
|
+
- Project-owned rules for what Codex may run, how results are checked, and when
|
|
49
|
+
a human must step in.
|
|
50
|
+
- Adaptive Acceptance Proof for runner-owned verification of UI, API, worker,
|
|
51
|
+
CLI, browser, mobile, and live-smoke behavior before draft PR handoff.
|
|
52
|
+
- Logs, summaries, and proof artifacts when available.
|
|
53
|
+
- Recovery for interrupted runner handoff when Codex finished locally but the
|
|
54
|
+
draft PR was not created yet.
|
|
50
55
|
- Draft PR handoff by default. No auto-merge.
|
|
51
56
|
|
|
52
57
|
## How It Works
|
|
53
58
|
|
|
54
|
-
|
|
59
|
+
At a high level, GitHub Issues are the queue, labels authorize work, isolated
|
|
60
|
+
worktrees keep runs separate, and the runner owns validation and publication.
|
|
61
|
+
|
|
62
|
+
```mermaid
|
|
63
|
+
flowchart TD
|
|
64
|
+
A["Target repo"] --> B["codex-orchestrator setup"]
|
|
65
|
+
B --> C[".codex-orchestrator/config.json + prompts"]
|
|
66
|
+
C --> D["GitHub Issue gets agent:auto or agent:plan-auto"]
|
|
67
|
+
D --> E["status / daemon / run"]
|
|
68
|
+
E --> R{"Recover interrupted handoff?"}
|
|
69
|
+
R -- "yes" --> L
|
|
70
|
+
R -- "no" --> F{"Eligible?"}
|
|
71
|
+
F -- "no" --> G["Skipped with reason"]
|
|
72
|
+
F -- "yes" --> H["Runner claims issue: agent:running"]
|
|
73
|
+
H --> I["Create isolated branch + worktree"]
|
|
74
|
+
I --> J["Build Codex prompt from issue + repo policy"]
|
|
75
|
+
J --> K["Run Codex CLI"]
|
|
76
|
+
K --> AAP{"Adaptive Acceptance Proof required?"}
|
|
77
|
+
AAP -- "no" --> L["Runner validates the full changeset"]
|
|
78
|
+
AAP -- "yes" --> AP["Run proof phase and collect artifacts"]
|
|
79
|
+
AP --> APR{"Proof result"}
|
|
80
|
+
APR -- "passed" --> L
|
|
81
|
+
APR -- "needs rework" --> J
|
|
82
|
+
APR -- "blocked" --> N
|
|
83
|
+
L --> M{"Gates pass?"}
|
|
84
|
+
M -- "no" --> N["Mark blocked, preserve evidence"]
|
|
85
|
+
M -- "yes" --> O["Push branch"]
|
|
86
|
+
O --> P["Open draft PR"]
|
|
87
|
+
P --> Q["Move issue to agent:review + post report"]
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
The important boundary is simple: Codex writes code, but the runner decides
|
|
91
|
+
whether that code can be handed to humans. The runner owns checks, acceptance
|
|
92
|
+
proof, labels, comments, branch pushes, and draft PR creation.
|
|
55
93
|
|
|
56
|
-
###
|
|
94
|
+
### Adaptive Acceptance Proof
|
|
95
|
+
|
|
96
|
+
Adaptive Acceptance Proof is the runner-owned verification phase for work that
|
|
97
|
+
needs observable product proof. After implementation, the runner can start a
|
|
98
|
+
separate proof phase that inspects the issue, changed files, and acceptance
|
|
99
|
+
criteria; runs focused browser, mobile, API, worker, CLI, or live-smoke checks;
|
|
100
|
+
and writes a machine-readable proof report with artifact links.
|
|
101
|
+
|
|
102
|
+
A result can reach draft PR handoff only when every required criterion maps to
|
|
103
|
+
high-confidence artifact evidence. If proof finds missing behavior, it returns a
|
|
104
|
+
concrete rework request and the runner loops back through implementation within
|
|
105
|
+
the configured iteration limit. If proof is malformed, low-confidence, lacks
|
|
106
|
+
artifacts, or changes product code during verification, the runner blocks
|
|
107
|
+
publication and preserves the evidence.
|
|
108
|
+
|
|
109
|
+
For UI proof, screenshots and UI dumps must also satisfy the UI Evidence
|
|
110
|
+
Contract: exact workflow, viewport coverage, current artifact freshness, layout
|
|
111
|
+
review, copy review, and source inputs. Screenshot-only proof cannot pass.
|
|
57
112
|
|
|
58
|
-
|
|
59
|
-
for child issues created by `agent:plan-auto`; those are marked with
|
|
60
|
-
`agent:child` and are executed only by the parent issue-tree flow.
|
|
113
|
+
There are two main ways to run work.
|
|
61
114
|
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
same-batch issues must not overlap by exact path or supported glob. Issues
|
|
66
|
-
without ownership metadata still run, but only one at a time.
|
|
115
|
+
### `agent:auto`
|
|
116
|
+
|
|
117
|
+
Use `agent:auto` for one clear standalone implementation issue.
|
|
67
118
|
|
|
68
119
|
The runner:
|
|
69
120
|
|
|
@@ -75,18 +126,26 @@ The runner:
|
|
|
75
126
|
6. Pushes the branch and opens a draft PR only after the gates pass.
|
|
76
127
|
7. Moves the issue to review and posts the run report.
|
|
77
128
|
|
|
129
|
+
When the daemon runs more than one `agent:auto` issue at once, it only batches
|
|
130
|
+
issues whose declared ownership does not overlap. Issues without ownership
|
|
131
|
+
metadata still run, but conservatively.
|
|
132
|
+
|
|
78
133
|
### `agent:plan-auto`
|
|
79
134
|
|
|
80
|
-
Use `agent:plan-auto` for work that
|
|
135
|
+
Use `agent:plan-auto` for larger work that should be planned before
|
|
136
|
+
implementation.
|
|
81
137
|
|
|
82
138
|
The runner asks Codex to plan the parent issue, break it into child issues,
|
|
83
|
-
|
|
84
|
-
integration draft
|
|
139
|
+
run safe children in dependency order, merge successful child branches into one
|
|
140
|
+
integration branch, validate that integration branch, and then open one draft
|
|
141
|
+
PR.
|
|
142
|
+
|
|
143
|
+
Child issues created by this flow use `agent:child`, not `agent:auto`. They are
|
|
144
|
+
owned by the parent run and are not picked up as standalone daemon work.
|
|
85
145
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
parent-owned child work with standalone scoped work.
|
|
146
|
+
If a runner stops after Codex finished locally but before draft PR handoff, the
|
|
147
|
+
runner can recover from its local state and completed report without rerunning
|
|
148
|
+
Codex. See [docs/deep-dive.md](docs/deep-dive.md) for the recovery rules.
|
|
90
149
|
|
|
91
150
|
## Basic Workflow
|
|
92
151
|
|
|
@@ -100,7 +159,13 @@ parent-owned child work with standalone scoped work.
|
|
|
100
159
|
|
|
101
160
|
The runner never auto-merges.
|
|
102
161
|
|
|
103
|
-
##
|
|
162
|
+
## Agent Memory
|
|
163
|
+
|
|
164
|
+
Repo-local Dreaming-lite memory lives in `docs/agents/memory/`. It is a small
|
|
165
|
+
curated lessons cache for repeated runner/debug/agent-workflow patterns, not a
|
|
166
|
+
replacement for `AGENTS.md`, ADRs, `docs/deep-dive.md`, or package prompts.
|
|
167
|
+
|
|
168
|
+
## Install
|
|
104
169
|
|
|
105
170
|
Requirements:
|
|
106
171
|
|
|
@@ -129,7 +194,7 @@ You can also run it with `npx`:
|
|
|
129
194
|
npx codex-orchestrator --help
|
|
130
195
|
```
|
|
131
196
|
|
|
132
|
-
##
|
|
197
|
+
## Set Up A Repository
|
|
133
198
|
|
|
134
199
|
Open the repository that should receive autonomous Codex work:
|
|
135
200
|
|
|
@@ -143,19 +208,19 @@ Run setup and create missing labels:
|
|
|
143
208
|
codex-orchestrator setup --prepare-labels
|
|
144
209
|
```
|
|
145
210
|
|
|
146
|
-
By default, setup reads the GitHub owner and repository name from `git remote
|
|
147
|
-
origin` and uses the current directory as the target repository. Use `--target`,
|
|
148
|
-
`--github-owner`, and `--github-repo` only when you need to override those
|
|
149
|
-
defaults.
|
|
150
|
-
|
|
151
211
|
Commit the generated `.codex-orchestrator/` directory to your repository. It is
|
|
152
212
|
the repository-owned policy for how autonomous work should run.
|
|
153
213
|
|
|
154
|
-
|
|
214
|
+
By default, setup reads the GitHub owner and repo from `git remote origin`. Use
|
|
215
|
+
`--target`, `--github-owner`, or `--github-repo` only when you need to override
|
|
216
|
+
that.
|
|
217
|
+
|
|
218
|
+
## Run Work
|
|
219
|
+
|
|
220
|
+
Check what the runner can see:
|
|
155
221
|
|
|
156
222
|
```sh
|
|
157
223
|
codex-orchestrator status --target .
|
|
158
|
-
codex-orchestrator status --target . --json
|
|
159
224
|
codex-orchestrator doctor --target .
|
|
160
225
|
```
|
|
161
226
|
|
|
@@ -171,12 +236,16 @@ Run the daemon:
|
|
|
171
236
|
codex-orchestrator daemon --target .
|
|
172
237
|
```
|
|
173
238
|
|
|
174
|
-
Run up to three independent scoped issues
|
|
239
|
+
Run up to three independent scoped issues at once:
|
|
175
240
|
|
|
176
241
|
```sh
|
|
177
242
|
codex-orchestrator daemon --target . --concurrency 3
|
|
178
243
|
```
|
|
179
244
|
|
|
245
|
+
`status` and `doctor` are read-only. `run` executes one selected issue.
|
|
246
|
+
`daemon` polls for eligible work and starts safe runs according to the policy in
|
|
247
|
+
`.codex-orchestrator/config.json`.
|
|
248
|
+
|
|
180
249
|
## Agent-Assisted Setup
|
|
181
250
|
|
|
182
251
|
You do not need a long prompt. You can ask an agent:
|
|
@@ -205,126 +274,60 @@ working in the repository can find repository-local setup guidance.
|
|
|
205
274
|
Use `--dry-run` only when you want a preview without writing files or creating
|
|
206
275
|
labels.
|
|
207
276
|
|
|
208
|
-
##
|
|
277
|
+
## What The Runner Checks
|
|
209
278
|
|
|
210
|
-
|
|
279
|
+
Before a result becomes a draft PR, the runner checks the whole local result:
|
|
211
280
|
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
281
|
+
- committed changes;
|
|
282
|
+
- staged changes;
|
|
283
|
+
- unstaged changes;
|
|
284
|
+
- untracked files;
|
|
285
|
+
- the completion report Codex was required to write;
|
|
286
|
+
- configured commands such as tests or type checks;
|
|
287
|
+
- review gates such as TDD evidence, changed tests, cleanup review, code review,
|
|
288
|
+
or acceptance proof when enabled;
|
|
289
|
+
- blocked paths and unsafe actions.
|
|
215
290
|
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
and the prompts used for planning and implementation.
|
|
220
|
-
|
|
221
|
-
The package ships bundled workflow prompts, so a repository does not need local
|
|
222
|
-
Codex `SKILL.md` files installed on the user's machine. Setup copies those
|
|
223
|
-
prompts into `.codex-orchestrator/prompts/workflows/`, and the runner reads the
|
|
224
|
-
copied prompt files during `agent:auto` and `agent:plan-auto` runs. Workflow
|
|
225
|
-
`skillName` values in config are descriptive metadata for the workflow role;
|
|
226
|
-
they are not a runtime dependency on the user's local Codex skill directory.
|
|
227
|
-
Setup also writes `.codex-orchestrator/prompts/manifest.json`, which lets later
|
|
228
|
-
setup runs tell apart untouched package prompts from prompts edited by the
|
|
229
|
-
project. By default, setup refreshes untouched prompts and reports conflicts for
|
|
230
|
-
locally edited prompts.
|
|
231
|
-
|
|
232
|
-
Configured checks run before publication. By default, missing
|
|
233
|
-
`npm run <script>` checks are reported as skipped warnings, not failures. You can
|
|
234
|
-
change that with `checksPolicy.missingNpmScript`.
|
|
235
|
-
|
|
236
|
-
For repos with existing lint debt, `checksPolicy.lintBaseline.mode` can be set
|
|
237
|
-
to `touched-only`. That lets a repo-wide lint failure be downgraded when a
|
|
238
|
-
separate touched-files lint command passes.
|
|
239
|
-
|
|
240
|
-
The default quality gate is conservative for runtime code changes. It can
|
|
241
|
-
require TDD evidence, changed tests, code review, cleanup review for larger
|
|
242
|
-
changes, and visual proof for UI work.
|
|
243
|
-
|
|
244
|
-
## Diagnostics
|
|
245
|
-
|
|
246
|
-
`doctor` is a read-only readiness check for operators. It validates the target
|
|
247
|
-
config, GitHub label visibility, git/base branch access, runner state paths,
|
|
248
|
-
configured checks, the Codex command, phase profiles, and visual proof settings.
|
|
249
|
-
It never launches Codex, creates worktrees, edits labels, or changes issues.
|
|
291
|
+
If the result passes, the runner pushes the branch and opens a draft PR. If it
|
|
292
|
+
does not pass, the runner marks the issue blocked, keeps the useful local
|
|
293
|
+
evidence, and explains what needs attention.
|
|
250
294
|
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
295
|
+
Acceptance proof is runner-owned. Codex can change product behavior, but the
|
|
296
|
+
runner runs proof afterwards and attaches screenshots, UI dumps, logs, smoke
|
|
297
|
+
outputs, or other artifacts to the PR and issue report. The proof phase must
|
|
298
|
+
produce a structured report that maps each required criterion to high-confidence
|
|
299
|
+
evidence. UI artifacts must include UI Evidence Contract mapping, and legacy
|
|
300
|
+
visual proof config only supplies migration inputs for report-producing proof.
|
|
255
301
|
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
not raw Codex transcripts, secrets, prompt text, or full issue comments.
|
|
260
|
-
|
|
261
|
-
Codex command profiles can be set per runner phase under `codex.profiles`.
|
|
262
|
-
Supported phases are `plan-parent`, `scoped-issue`, `tree-child`,
|
|
263
|
-
`fresh-context-review`, `visual-proof`, and `quality-review`. Missing profile
|
|
264
|
-
fields fall back to the global `codex.command`, `codex.args`, `timeoutMs`, and
|
|
265
|
-
`idleTimeoutMs`, so existing configs keep working.
|
|
266
|
-
|
|
267
|
-
Each Codex session writes a bounded context snapshot before invocation and links
|
|
268
|
-
it from lifecycle events under the runner state directory. Snapshots record the
|
|
269
|
-
issue identity, runner decision, selected profile, workspace paths, and
|
|
270
|
-
publication boundaries so a maintainer can reproduce why a session started
|
|
271
|
-
without reading raw logs.
|
|
272
|
-
|
|
273
|
-
## Visual Proof
|
|
274
|
-
|
|
275
|
-
For browser UI work, configure a runner-owned proof command, usually a
|
|
276
|
-
Playwright script:
|
|
277
|
-
|
|
278
|
-
```json
|
|
279
|
-
{
|
|
280
|
-
"reviewGates": {
|
|
281
|
-
"visualProof": {
|
|
282
|
-
"runnerValidationCommand": "npm run visual-proof -- --issue ${issueNumber}",
|
|
283
|
-
"runnerTimeoutMs": 900000,
|
|
284
|
-
"envPassthrough": [
|
|
285
|
-
"CODEX_ORCHESTRATOR_LOGIN_EMAIL",
|
|
286
|
-
"CODEX_ORCHESTRATOR_LOGIN_PASSWORD"
|
|
287
|
-
]
|
|
288
|
-
}
|
|
289
|
-
}
|
|
290
|
-
}
|
|
291
|
-
```
|
|
302
|
+
## Repository Policy
|
|
303
|
+
|
|
304
|
+
Every installed repository owns its automation policy in:
|
|
292
305
|
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
worktree path, and changed files.
|
|
306
|
+
```sh
|
|
307
|
+
.codex-orchestrator/config.json
|
|
308
|
+
```
|
|
297
309
|
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
310
|
+
That config controls labels, branch names, checks, review gates, blocked paths,
|
|
311
|
+
prompt files, child concurrency, and PR titles. The package provides defaults;
|
|
312
|
+
the target repository decides how strict they should be.
|
|
301
313
|
|
|
302
|
-
For
|
|
303
|
-
|
|
304
|
-
no usable device is reported as a warning with the concrete reason, not as an
|
|
305
|
-
automatic release blocker. Native Android proof uses the project Gradle wrapper
|
|
306
|
-
with a writable Gradle cache; Flutter-specific SDK cache recovery is used only
|
|
307
|
-
for Flutter projects and only through a preconfigured writable SDK path in
|
|
308
|
-
`CODEX_ORCHESTRATOR_FLUTTER_ROOT`. Native iOS proof uses Xcode simulator/device
|
|
309
|
-
tooling with a writable DerivedData path.
|
|
314
|
+
For the full config surface and technical behavior, see
|
|
315
|
+
[docs/deep-dive.md](docs/deep-dive.md).
|
|
310
316
|
|
|
311
317
|
## Safety Model
|
|
312
318
|
|
|
313
319
|
The package is PR-first and human-reviewed. The important guardrails are:
|
|
314
320
|
|
|
315
|
-
- no automatic merge
|
|
321
|
+
- no automatic merge;
|
|
322
|
+
- draft PRs only;
|
|
316
323
|
- Codex may change files, but the runner owns remote publication and GitHub
|
|
317
|
-
state;
|
|
324
|
+
state changes;
|
|
318
325
|
- only explicitly authorized issues run;
|
|
319
326
|
- child issues are never inferred from ordinary links or references;
|
|
320
|
-
- committed and uncommitted changes are checked
|
|
321
|
-
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
- bounded rework stops at the configured limit;
|
|
325
|
-
- Policy Suggestions are recommendations only;
|
|
326
|
-
- underspecified work can be blocked for maintainer clarification instead of
|
|
327
|
-
letting Codex invent product decisions.
|
|
327
|
+
- committed and uncommitted changes are checked;
|
|
328
|
+
- missing or malformed completion reports block publication;
|
|
329
|
+
- secret files, destructive data/cache actions, and production deploy or release
|
|
330
|
+
actions are blocked by default.
|
|
328
331
|
|
|
329
332
|
## Labels
|
|
330
333
|
|
|
@@ -352,47 +355,15 @@ codex-orchestrator setup [--target <path>] [--github-owner <owner>] \
|
|
|
352
355
|
[--github-repo <repo>] [--dry-run] [--prepare-labels]
|
|
353
356
|
codex-orchestrator status --target <path> [--dry-run] [--json]
|
|
354
357
|
codex-orchestrator run --target <path> --issue <number>
|
|
358
|
+
codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
|
|
359
|
+
codex-orchestrator visual-proof android --issue <number> [--target <path>]
|
|
360
|
+
codex-orchestrator visual-proof ios --issue <number> [--target <path>]
|
|
355
361
|
codex-orchestrator daemon --target <path> [--once] \
|
|
356
362
|
[--interval-seconds <seconds>] [--max-runs <count>] \
|
|
357
363
|
[--concurrency <count>]
|
|
358
364
|
```
|
|
359
365
|
|
|
360
|
-
`
|
|
361
|
-
`.codex-orchestrator/`. Useful flags:
|
|
362
|
-
|
|
363
|
-
- `--dry-run` - show the setup plan without writing files or creating labels;
|
|
364
|
-
- `--prepare-labels` - create missing GitHub labels;
|
|
365
|
-
- `--target <path>` - override the target directory, which defaults to the current directory;
|
|
366
|
-
- `--github-owner <owner>` - override the GitHub owner inferred from `origin`;
|
|
367
|
-
- `--github-repo <repo>` - override the GitHub repo inferred from `origin`;
|
|
368
|
-
- `--sync-prompts <auto|keep|replace|merge>` - choose how package-bundled
|
|
369
|
-
prompt updates are applied. `auto` refreshes untouched prompts and reports
|
|
370
|
-
local-edit conflicts, `keep` preserves existing prompts, `replace` overwrites
|
|
371
|
-
with bundled prompts, and `merge` appends bundled updates to locally edited
|
|
372
|
-
prompts;
|
|
373
|
-
- `--replace-package-skills` - refresh package-bundled prompt files. The flag
|
|
374
|
-
name is kept for compatibility and behaves like `--sync-prompts=replace`; it
|
|
375
|
-
does not install or require local Codex skills.
|
|
376
|
-
|
|
377
|
-
Setup does not launch Codex, commit changes, or open pull requests.
|
|
378
|
-
When the target repository already has a `package.json`, setup also adds
|
|
379
|
-
`orchestrator:*` npm scripts. Daemon scripts run `doctor` first, then start the
|
|
380
|
-
daemon only if the readiness check passes.
|
|
381
|
-
|
|
382
|
-
`status` is read-only. It shows eligible issues, skipped issues with reasons,
|
|
383
|
-
and local recovery state.
|
|
384
|
-
|
|
385
|
-
`run` executes one selected issue when labels and state allow it. `agent:auto`
|
|
386
|
-
opens one scoped draft PR. `agent:plan-auto` runs parent planning, child waves,
|
|
387
|
-
final validation, and one integration draft PR.
|
|
388
|
-
|
|
389
|
-
`daemon` polls for eligible work. By default, fresh setup config allows up to
|
|
390
|
-
three scoped issues per batch through `runner.maxParallelScopedIssues`; legacy
|
|
391
|
-
configs without that field remain sequential unless `--concurrency` is passed.
|
|
392
|
-
Only scoped issues with non-overlapping ownership metadata can share a batch.
|
|
393
|
-
`agent:plan-auto` runs remain exclusive. The daemon also cleans up runner-owned
|
|
394
|
-
worktrees after their PRs are merged, while preserving dirty, blocked, active,
|
|
395
|
-
or unpublished worktrees for inspection.
|
|
366
|
+
Use `codex-orchestrator <command> --help` for command-specific flags.
|
|
396
367
|
|
|
397
368
|
## Current Scope
|
|
398
369
|
|
package/dist/src/cli.js
CHANGED
|
@@ -8,7 +8,11 @@ import { runDaemonCommand } from './runner/daemon-command.js';
|
|
|
8
8
|
import { runDoctorCommand } from './runner/doctor-command.js';
|
|
9
9
|
import { runPlanAutoCommand } from './runner/plan-auto-command.js';
|
|
10
10
|
import { runScopedAutoCommand } from './runner/scoped-auto-command.js';
|
|
11
|
+
import { recoverScopedRun } from './runner/scoped-recovery.js';
|
|
11
12
|
import { runStatusCommand } from './runner/status-command.js';
|
|
13
|
+
import { parseAndroidVisualProofArgs, runAndroidVisualProofCommand } from './runner/android-visual-proof-command.js';
|
|
14
|
+
import { parseIosVisualProofArgs, runIosVisualProofCommand } from './runner/ios-visual-proof-command.js';
|
|
15
|
+
import { parseMobileVisualProofArgs, runMobileVisualProofCommand } from './runner/mobile-visual-proof-command.js';
|
|
12
16
|
import { runSetupCommand } from './setup/setup-command.js';
|
|
13
17
|
import { promptSyncModes } from './setup/prompt-sync.js';
|
|
14
18
|
const helpText = `codex-orchestrator
|
|
@@ -22,6 +26,9 @@ Usage:
|
|
|
22
26
|
codex-orchestrator status --target <path> [--dry-run] [--json]
|
|
23
27
|
codex-orchestrator run --target <path> --issue <number>
|
|
24
28
|
codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>] [--concurrency <number>]
|
|
29
|
+
codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
|
|
30
|
+
codex-orchestrator visual-proof android --issue <number> [--target <path>]
|
|
31
|
+
codex-orchestrator visual-proof ios --issue <number> [--target <path>]
|
|
25
32
|
|
|
26
33
|
Commands:
|
|
27
34
|
health Run a no-op local health check.
|
|
@@ -30,6 +37,7 @@ Commands:
|
|
|
30
37
|
status Show eligible/skipped issue work and local recovery state.
|
|
31
38
|
run Execute one authorized issue: scoped agent:auto or full agent:plan-auto issue tree.
|
|
32
39
|
daemon Poll GitHub Issues and execute eligible autonomous work until stopped.
|
|
40
|
+
visual-proof Run package-owned proof commands used by review gates.
|
|
33
41
|
|
|
34
42
|
Options:
|
|
35
43
|
--help, -h Show this help.
|
|
@@ -164,6 +172,48 @@ async function main(args) {
|
|
|
164
172
|
return 1;
|
|
165
173
|
}
|
|
166
174
|
}
|
|
175
|
+
if (command === 'visual-proof') {
|
|
176
|
+
const [kind, ...rest] = args.slice(1);
|
|
177
|
+
if (kind !== 'mobile' && kind !== 'android' && kind !== 'ios') {
|
|
178
|
+
process.stderr.write('visual-proof requires a supported kind: mobile, android, or ios\nRun codex-orchestrator --help for usage.\n');
|
|
179
|
+
return 2;
|
|
180
|
+
}
|
|
181
|
+
try {
|
|
182
|
+
if (kind === 'mobile') {
|
|
183
|
+
const parsed = parseMobileVisualProofArgs(rest);
|
|
184
|
+
if (!parsed.ok) {
|
|
185
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
186
|
+
return 2;
|
|
187
|
+
}
|
|
188
|
+
await runMobileVisualProofCommand(parsed.value);
|
|
189
|
+
process.stdout.write(`mobile visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
190
|
+
}
|
|
191
|
+
else if (kind === 'ios') {
|
|
192
|
+
const parsed = parseIosVisualProofArgs(rest);
|
|
193
|
+
if (!parsed.ok) {
|
|
194
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
195
|
+
return 2;
|
|
196
|
+
}
|
|
197
|
+
await runIosVisualProofCommand(parsed.value);
|
|
198
|
+
process.stdout.write(`ios visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
199
|
+
}
|
|
200
|
+
else {
|
|
201
|
+
const parsed = parseAndroidVisualProofArgs(rest);
|
|
202
|
+
if (!parsed.ok) {
|
|
203
|
+
process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
|
|
204
|
+
return 2;
|
|
205
|
+
}
|
|
206
|
+
await runAndroidVisualProofCommand(parsed.value);
|
|
207
|
+
process.stdout.write(`android visual proof captured for issue #${parsed.value.issueNumber}\n`);
|
|
208
|
+
}
|
|
209
|
+
return 0;
|
|
210
|
+
}
|
|
211
|
+
catch (error) {
|
|
212
|
+
const message = error instanceof Error ? error.message : 'visual proof failed';
|
|
213
|
+
process.stderr.write(`${message}\n`);
|
|
214
|
+
return 1;
|
|
215
|
+
}
|
|
216
|
+
}
|
|
167
217
|
process.stderr.write(`Unknown command: ${command}\nRun codex-orchestrator --help for usage.\n`);
|
|
168
218
|
return 1;
|
|
169
219
|
}
|
|
@@ -177,6 +227,15 @@ async function runIssueCommand(targetRootInput, issueNumber) {
|
|
|
177
227
|
}
|
|
178
228
|
const decision = discoverIssueWork([issue], config)[0];
|
|
179
229
|
if (!decision || decision.kind !== 'eligible') {
|
|
230
|
+
const recovered = await recoverScopedRun({
|
|
231
|
+
targetRoot,
|
|
232
|
+
issueNumber,
|
|
233
|
+
invocation: 'targeted',
|
|
234
|
+
issueAdapter,
|
|
235
|
+
});
|
|
236
|
+
if (recovered.status !== 'not-recoverable') {
|
|
237
|
+
return { reportComment: recovered.reportComment };
|
|
238
|
+
}
|
|
180
239
|
const reason = decision?.kind === 'skipped' ? decision.reason : 'not eligible';
|
|
181
240
|
throw new Error(`Issue #${issueNumber} is not eligible for autonomous work: ${reason}`);
|
|
182
241
|
}
|