codex-orchestrator 0.1.31 → 0.1.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/README.md +145 -174
  3. package/dist/src/cli.js +59 -0
  4. package/dist/src/cli.js.map +1 -1
  5. package/dist/src/codex/command-adapter.d.ts +2 -1
  6. package/dist/src/codex/command-adapter.d.ts.map +1 -1
  7. package/dist/src/codex/command-adapter.js +13 -2
  8. package/dist/src/codex/command-adapter.js.map +1 -1
  9. package/dist/src/codex/mobile-device-guard.d.ts +7 -0
  10. package/dist/src/codex/mobile-device-guard.d.ts.map +1 -0
  11. package/dist/src/codex/mobile-device-guard.js +63 -0
  12. package/dist/src/codex/mobile-device-guard.js.map +1 -0
  13. package/dist/src/config/constants.d.ts +1 -1
  14. package/dist/src/config/constants.d.ts.map +1 -1
  15. package/dist/src/config/constants.js +1 -0
  16. package/dist/src/config/constants.js.map +1 -1
  17. package/dist/src/config/schema.d.ts +15 -2
  18. package/dist/src/config/schema.d.ts.map +1 -1
  19. package/dist/src/config/schema.js +30 -0
  20. package/dist/src/config/schema.js.map +1 -1
  21. package/dist/src/github/gh-pull-request-adapter.d.ts +1 -0
  22. package/dist/src/github/gh-pull-request-adapter.d.ts.map +1 -1
  23. package/dist/src/github/gh-pull-request-adapter.js +20 -0
  24. package/dist/src/github/gh-pull-request-adapter.js.map +1 -1
  25. package/dist/src/github/pull-requests.d.ts +2 -0
  26. package/dist/src/github/pull-requests.d.ts.map +1 -1
  27. package/dist/src/github/pull-requests.js +9 -0
  28. package/dist/src/github/pull-requests.js.map +1 -1
  29. package/dist/src/runner/acceptance-proof-runner.d.ts +54 -0
  30. package/dist/src/runner/acceptance-proof-runner.d.ts.map +1 -0
  31. package/dist/src/runner/acceptance-proof-runner.js +229 -0
  32. package/dist/src/runner/acceptance-proof-runner.js.map +1 -0
  33. package/dist/src/runner/acceptance-proof.d.ts +121 -0
  34. package/dist/src/runner/acceptance-proof.d.ts.map +1 -0
  35. package/dist/src/runner/acceptance-proof.js +405 -0
  36. package/dist/src/runner/acceptance-proof.js.map +1 -0
  37. package/dist/src/runner/android-visual-proof-command.d.ts +26 -0
  38. package/dist/src/runner/android-visual-proof-command.d.ts.map +1 -0
  39. package/dist/src/runner/android-visual-proof-command.js +598 -0
  40. package/dist/src/runner/android-visual-proof-command.js.map +1 -0
  41. package/dist/src/runner/command-utils.d.ts.map +1 -1
  42. package/dist/src/runner/command-utils.js +55 -6
  43. package/dist/src/runner/command-utils.js.map +1 -1
  44. package/dist/src/runner/completion-report.d.ts +3 -1
  45. package/dist/src/runner/completion-report.d.ts.map +1 -1
  46. package/dist/src/runner/completion-report.js +2 -1
  47. package/dist/src/runner/completion-report.js.map +1 -1
  48. package/dist/src/runner/daemon-command.d.ts.map +1 -1
  49. package/dist/src/runner/daemon-command.js +20 -0
  50. package/dist/src/runner/daemon-command.js.map +1 -1
  51. package/dist/src/runner/doctor-command.js +3 -3
  52. package/dist/src/runner/doctor-command.js.map +1 -1
  53. package/dist/src/runner/durable-run-summary.d.ts +3 -0
  54. package/dist/src/runner/durable-run-summary.d.ts.map +1 -1
  55. package/dist/src/runner/durable-run-summary.js +2 -0
  56. package/dist/src/runner/durable-run-summary.js.map +1 -1
  57. package/dist/src/runner/handoff-evidence.d.ts +5 -0
  58. package/dist/src/runner/handoff-evidence.d.ts.map +1 -1
  59. package/dist/src/runner/handoff-evidence.js +22 -0
  60. package/dist/src/runner/handoff-evidence.js.map +1 -1
  61. package/dist/src/runner/ios-visual-proof-command.d.ts +25 -0
  62. package/dist/src/runner/ios-visual-proof-command.d.ts.map +1 -0
  63. package/dist/src/runner/ios-visual-proof-command.js +355 -0
  64. package/dist/src/runner/ios-visual-proof-command.js.map +1 -0
  65. package/dist/src/runner/lifecycle-events.d.ts +1 -1
  66. package/dist/src/runner/lifecycle-events.d.ts.map +1 -1
  67. package/dist/src/runner/local-execution-session.d.ts +19 -6
  68. package/dist/src/runner/local-execution-session.d.ts.map +1 -1
  69. package/dist/src/runner/local-execution-session.js +93 -11
  70. package/dist/src/runner/local-execution-session.js.map +1 -1
  71. package/dist/src/runner/local-state.d.ts +6 -0
  72. package/dist/src/runner/local-state.d.ts.map +1 -1
  73. package/dist/src/runner/local-state.js +14 -0
  74. package/dist/src/runner/local-state.js.map +1 -1
  75. package/dist/src/runner/mobile-device-lease.d.ts +11 -0
  76. package/dist/src/runner/mobile-device-lease.d.ts.map +1 -0
  77. package/dist/src/runner/mobile-device-lease.js +121 -0
  78. package/dist/src/runner/mobile-device-lease.js.map +1 -0
  79. package/dist/src/runner/mobile-visual-proof-command.d.ts +17 -0
  80. package/dist/src/runner/mobile-visual-proof-command.d.ts.map +1 -0
  81. package/dist/src/runner/mobile-visual-proof-command.js +168 -0
  82. package/dist/src/runner/mobile-visual-proof-command.js.map +1 -0
  83. package/dist/src/runner/plan-auto-command.d.ts.map +1 -1
  84. package/dist/src/runner/plan-auto-command.js +57 -6
  85. package/dist/src/runner/plan-auto-command.js.map +1 -1
  86. package/dist/src/runner/prompt.js +2 -2
  87. package/dist/src/runner/prompt.js.map +1 -1
  88. package/dist/src/runner/recovery.d.ts +2 -1
  89. package/dist/src/runner/recovery.d.ts.map +1 -1
  90. package/dist/src/runner/recovery.js +24 -0
  91. package/dist/src/runner/recovery.js.map +1 -1
  92. package/dist/src/runner/review-gate-policy.d.ts +7 -0
  93. package/dist/src/runner/review-gate-policy.d.ts.map +1 -1
  94. package/dist/src/runner/review-gate-policy.js +81 -20
  95. package/dist/src/runner/review-gate-policy.js.map +1 -1
  96. package/dist/src/runner/review-gates.d.ts.map +1 -1
  97. package/dist/src/runner/review-gates.js +44 -11
  98. package/dist/src/runner/review-gates.js.map +1 -1
  99. package/dist/src/runner/rework-policy.d.ts +1 -0
  100. package/dist/src/runner/rework-policy.d.ts.map +1 -1
  101. package/dist/src/runner/rework-policy.js +7 -0
  102. package/dist/src/runner/rework-policy.js.map +1 -1
  103. package/dist/src/runner/scoped-auto-command.d.ts +46 -1
  104. package/dist/src/runner/scoped-auto-command.d.ts.map +1 -1
  105. package/dist/src/runner/scoped-auto-command.js +147 -47
  106. package/dist/src/runner/scoped-auto-command.js.map +1 -1
  107. package/dist/src/runner/scoped-recovery.d.ts +65 -0
  108. package/dist/src/runner/scoped-recovery.d.ts.map +1 -0
  109. package/dist/src/runner/scoped-recovery.js +484 -0
  110. package/dist/src/runner/scoped-recovery.js.map +1 -0
  111. package/dist/src/runner/status-command.d.ts.map +1 -1
  112. package/dist/src/runner/status-command.js +1 -0
  113. package/dist/src/runner/status-command.js.map +1 -1
  114. package/dist/src/runner/visual-proof-runner.d.ts +1 -0
  115. package/dist/src/runner/visual-proof-runner.d.ts.map +1 -1
  116. package/dist/src/runner/visual-proof-runner.js +144 -16
  117. package/dist/src/runner/visual-proof-runner.js.map +1 -1
  118. package/dist/src/setup/project-config.d.ts +4 -0
  119. package/dist/src/setup/project-config.d.ts.map +1 -1
  120. package/dist/src/setup/project-config.js +141 -9
  121. package/dist/src/setup/project-config.js.map +1 -1
  122. package/dist/src/setup/setup-command.d.ts.map +1 -1
  123. package/dist/src/setup/setup-command.js +2 -2
  124. package/dist/src/setup/setup-command.js.map +1 -1
  125. package/dist/src/setup/workflows.d.ts.map +1 -1
  126. package/dist/src/setup/workflows.js +5 -0
  127. package/dist/src/setup/workflows.js.map +1 -1
  128. package/docs/deep-dive.md +217 -28
  129. package/package.json +1 -1
  130. package/prompts/workflows/acceptance-proof.md +25 -0
  131. package/prompts/workflows/scoped-implementation.md +6 -2
package/CHANGELOG.md CHANGED
@@ -6,6 +6,71 @@ The format is based on Keep a Changelog, and this project follows SemVer.
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.35] - 2026-05-21
10
+
11
+ ### Added
12
+ - Added a runner-validated UI Evidence Contract for Acceptance Proof reports,
13
+ covering workflow, viewport, freshness, layout, copy, and source-input
14
+ evidence for screenshot and UI-dump artifacts.
15
+ - Added live smoke coverage for UI Evidence pass and blocking cases, including
16
+ missing UI Evidence and too-narrow desktop viewport proof.
17
+
18
+ ### Changed
19
+ - Runner-owned visual proof no longer treats screenshot-only command success as
20
+ a pass path; proof commands must produce a valid machine-readable Acceptance
21
+ Proof report.
22
+ - Updated the legacy `visual-proof` live smoke scenario to emit the same
23
+ machine-readable UI Evidence report required by the runner.
24
+
25
+ ## [0.1.34] - 2026-05-20
26
+
27
+ ### Added
28
+ - Added an Adaptive Proof Agent Codex phase for scoped and issue-tree child
29
+ runs, with runner-provided proof report paths, artifact directories, changed
30
+ file context, and proof-owned repair policy.
31
+ - Added durable Acceptance Proof attempt evidence in lifecycle events, run
32
+ summaries, blocked comments, review reports, and issue-tree PR handoff.
33
+ - Added package-bundled Acceptance Proof workflow prompts and setup routing for
34
+ the new proof phase.
35
+
36
+ ### Changed
37
+ - Parent `agent:plan-auto` child waves now block parent publication when a child
38
+ Acceptance Proof attempt fails, requests rework, or is blocked.
39
+ - Proof attempts now use isolated Codex homes and preserve proof artifacts while
40
+ keeping publication authority runner-owned.
41
+
42
+ ## [0.1.33] - 2026-05-20
43
+
44
+ ### Added
45
+ - Added canonical `reviewGates.acceptanceProof` policy with proof-owned path
46
+ classification and machine validation for high-confidence proof reports.
47
+ - Added live smoke scenarios for canonical Acceptance Proof pass, proof rework,
48
+ low-confidence blocking, and proof-phase product-diff blocking.
49
+
50
+ ### Changed
51
+ - Kept `reviewGates.visualProof` as a compatibility adapter while routing
52
+ runner prompts and proof policy through Acceptance Proof.
53
+ - Proof-phase product-code changes now block publishability instead of being
54
+ silently committed as verification output.
55
+
56
+ ## [0.1.32] - 2026-05-19
57
+
58
+ ### Added
59
+ - Added package-owned `visual-proof mobile`, `visual-proof android`, and
60
+ `visual-proof ios` commands for reusable UI launch proof across installed
61
+ repositories.
62
+ - Mobile visual proof now supports Flutter Android, native Android, Flutter iOS,
63
+ and native iOS projects, with screenshots saved into runner proof artifacts.
64
+
65
+ ### Changed
66
+ - Setup now defaults visual proof to
67
+ `codex-orchestrator visual-proof mobile --issue ${issueNumber}` instead of a
68
+ target-repo local proof script.
69
+ - Android proof resolves SDK tools from environment variables, `PATH`, and
70
+ default macOS, Linux, and Windows SDK locations.
71
+ - On macOS, mobile proof falls back to the iOS simulator when Android tooling or
72
+ devices are unavailable and the repo has an iOS target.
73
+
9
74
  ## [0.1.30] - 2026-05-18
10
75
 
11
76
  ### Added
package/README.md CHANGED
@@ -1,6 +1,10 @@
1
1
  # codex-orchestrator
2
2
 
3
- `codex-orchestrator` turns GitHub Issues into controlled Codex work.
3
+ ## About
4
+
5
+ `codex-orchestrator` turns GitHub Issues and project work into isolated,
6
+ autonomous Codex implementation runs, allowing maintainers to manage work
7
+ instead of supervising coding agents.
4
8
 
5
9
  Instead of starting a new Codex chat for every issue, you label the work you
6
10
  want automated. The runner creates an isolated workspace, gives Codex the issue
@@ -41,29 +45,76 @@ control to humans before anything is merged.
41
45
  - A repeatable way to send selected GitHub Issues to Codex.
42
46
  - One-off autonomous runs for scoped implementation tasks.
43
47
  - Parent planning for larger features, with child issues executed in safe waves.
44
- - Project-owned rules for labels, branches, prompts, checks, review gates, and
45
- blocked actions.
46
- - Full change-set checks, including local commits, staged files, unstaged files,
47
- and untracked files.
48
- - Durable logs and summaries when a run is interrupted, blocked, or ready for
49
- review.
48
+ - Project-owned rules for what Codex may run, how results are checked, and when
49
+ a human must step in.
50
+ - Adaptive Acceptance Proof for runner-owned verification of UI, API, worker,
51
+ CLI, browser, mobile, and live-smoke behavior before draft PR handoff.
52
+ - Logs, summaries, and proof artifacts when available.
53
+ - Recovery for interrupted runner handoff when Codex finished locally but the
54
+ draft PR was not created yet.
50
55
  - Draft PR handoff by default. No auto-merge.
51
56
 
52
57
  ## How It Works
53
58
 
54
- There are two main modes.
59
+ At a high level, GitHub Issues are the queue, labels authorize work, isolated
60
+ worktrees keep runs separate, and the runner owns validation and publication.
61
+
62
+ ```mermaid
63
+ flowchart TD
64
+ A["Target repo"] --> B["codex-orchestrator setup"]
65
+ B --> C[".codex-orchestrator/config.json + prompts"]
66
+ C --> D["GitHub Issue gets agent:auto or agent:plan-auto"]
67
+ D --> E["status / daemon / run"]
68
+ E --> R{"Recover interrupted handoff?"}
69
+ R -- "yes" --> L
70
+ R -- "no" --> F{"Eligible?"}
71
+ F -- "no" --> G["Skipped with reason"]
72
+ F -- "yes" --> H["Runner claims issue: agent:running"]
73
+ H --> I["Create isolated branch + worktree"]
74
+ I --> J["Build Codex prompt from issue + repo policy"]
75
+ J --> K["Run Codex CLI"]
76
+ K --> AAP{"Adaptive Acceptance Proof required?"}
77
+ AAP -- "no" --> L["Runner validates the full changeset"]
78
+ AAP -- "yes" --> AP["Run proof phase and collect artifacts"]
79
+ AP --> APR{"Proof result"}
80
+ APR -- "passed" --> L
81
+ APR -- "needs rework" --> J
82
+ APR -- "blocked" --> N
83
+ L --> M{"Gates pass?"}
84
+ M -- "no" --> N["Mark blocked, preserve evidence"]
85
+ M -- "yes" --> O["Push branch"]
86
+ O --> P["Open draft PR"]
87
+ P --> Q["Move issue to agent:review + post report"]
88
+ ```
89
+
90
+ The important boundary is simple: Codex writes code, but the runner decides
91
+ whether that code can be handed to humans. The runner owns checks, acceptance
92
+ proof, labels, comments, branch pushes, and draft PR creation.
55
93
 
56
- ### `agent:auto`
94
+ ### Adaptive Acceptance Proof
95
+
96
+ Adaptive Acceptance Proof is the runner-owned verification phase for work that
97
+ needs observable product proof. After implementation, the runner can start a
98
+ separate proof phase that inspects the issue, changed files, and acceptance
99
+ criteria; runs focused browser, mobile, API, worker, CLI, or live-smoke checks;
100
+ and writes a machine-readable proof report with artifact links.
101
+
102
+ A result can reach draft PR handoff only when every required criterion maps to
103
+ high-confidence artifact evidence. If proof finds missing behavior, it returns a
104
+ concrete rework request and the runner loops back through implementation within
105
+ the configured iteration limit. If proof is malformed, low-confidence, lacks
106
+ artifacts, or changes product code during verification, the runner blocks
107
+ publication and preserves the evidence.
108
+
109
+ For UI proof, screenshots and UI dumps must also satisfy the UI Evidence
110
+ Contract: exact workflow, viewport coverage, current artifact freshness, layout
111
+ review, copy review, and source inputs. Screenshot-only proof cannot pass.
57
112
 
58
- Use `agent:auto` for one clear standalone implementation issue. Do not use it
59
- for child issues created by `agent:plan-auto`; those are marked with
60
- `agent:child` and are executed only by the parent issue-tree flow.
113
+ There are two main ways to run work.
61
114
 
62
- When a daemon is allowed to run more than one scoped issue at a time, parallel
63
- `agent:auto` selection is conservative. An issue must include a
64
- `## codex-orchestrator metadata` section with an `Ownership:` bullet list, and
65
- same-batch issues must not overlap by exact path or supported glob. Issues
66
- without ownership metadata still run, but only one at a time.
115
+ ### `agent:auto`
116
+
117
+ Use `agent:auto` for one clear standalone implementation issue.
67
118
 
68
119
  The runner:
69
120
 
@@ -75,18 +126,26 @@ The runner:
75
126
  6. Pushes the branch and opens a draft PR only after the gates pass.
76
127
  7. Moves the issue to review and posts the run report.
77
128
 
129
+ When the daemon runs more than one `agent:auto` issue at once, it only batches
130
+ issues whose declared ownership does not overlap. Issues without ownership
131
+ metadata still run, but conservatively.
132
+
78
133
  ### `agent:plan-auto`
79
134
 
80
- Use `agent:plan-auto` for work that needs planning first.
135
+ Use `agent:plan-auto` for larger work that should be planned before
136
+ implementation.
81
137
 
82
138
  The runner asks Codex to plan the parent issue, break it into child issues,
83
- triage them, run safe children in dependency order, and then open one
84
- integration draft PR.
139
+ run safe children in dependency order, merge successful child branches into one
140
+ integration branch, validate that integration branch, and then open one draft
141
+ PR.
142
+
143
+ Child issues created by this flow use `agent:child`, not `agent:auto`. They are
144
+ owned by the parent run and are not picked up as standalone daemon work.
85
145
 
86
- Only child issues explicitly marked by the runner belong to the autonomous tree.
87
- Ordinary links, milestones, project fields, or casual references are not enough.
88
- Child issues use `agent:child`, not `agent:auto`, so the daemon cannot confuse
89
- parent-owned child work with standalone scoped work.
146
+ If a runner stops after Codex finished locally but before draft PR handoff, the
147
+ runner can recover from its local state and completed report without rerunning
148
+ Codex. See [docs/deep-dive.md](docs/deep-dive.md) for the recovery rules.
90
149
 
91
150
  ## Basic Workflow
92
151
 
@@ -100,7 +159,13 @@ parent-owned child work with standalone scoped work.
100
159
 
101
160
  The runner never auto-merges.
102
161
 
103
- ## Installation
162
+ ## Agent Memory
163
+
164
+ Repo-local Dreaming-lite memory lives in `docs/agents/memory/`. It is a small
165
+ curated lessons cache for repeated runner/debug/agent-workflow patterns, not a
166
+ replacement for `AGENTS.md`, ADRs, `docs/deep-dive.md`, or package prompts.
167
+
168
+ ## Install
104
169
 
105
170
  Requirements:
106
171
 
@@ -129,7 +194,7 @@ You can also run it with `npx`:
129
194
  npx codex-orchestrator --help
130
195
  ```
131
196
 
132
- ## Quick Start
197
+ ## Set Up A Repository
133
198
 
134
199
  Open the repository that should receive autonomous Codex work:
135
200
 
@@ -143,19 +208,19 @@ Run setup and create missing labels:
143
208
  codex-orchestrator setup --prepare-labels
144
209
  ```
145
210
 
146
- By default, setup reads the GitHub owner and repository name from `git remote
147
- origin` and uses the current directory as the target repository. Use `--target`,
148
- `--github-owner`, and `--github-repo` only when you need to override those
149
- defaults.
150
-
151
211
  Commit the generated `.codex-orchestrator/` directory to your repository. It is
152
212
  the repository-owned policy for how autonomous work should run.
153
213
 
154
- Check eligible work:
214
+ By default, setup reads the GitHub owner and repo from `git remote origin`. Use
215
+ `--target`, `--github-owner`, or `--github-repo` only when you need to override
216
+ that.
217
+
218
+ ## Run Work
219
+
220
+ Check what the runner can see:
155
221
 
156
222
  ```sh
157
223
  codex-orchestrator status --target .
158
- codex-orchestrator status --target . --json
159
224
  codex-orchestrator doctor --target .
160
225
  ```
161
226
 
@@ -171,12 +236,16 @@ Run the daemon:
171
236
  codex-orchestrator daemon --target .
172
237
  ```
173
238
 
174
- Run up to three independent scoped issues in one daemon batch:
239
+ Run up to three independent scoped issues at once:
175
240
 
176
241
  ```sh
177
242
  codex-orchestrator daemon --target . --concurrency 3
178
243
  ```
179
244
 
245
+ `status` and `doctor` are read-only. `run` executes one selected issue.
246
+ `daemon` polls for eligible work and starts safe runs according to the policy in
247
+ `.codex-orchestrator/config.json`.
248
+
180
249
  ## Agent-Assisted Setup
181
250
 
182
251
  You do not need a long prompt. You can ask an agent:
@@ -205,126 +274,60 @@ working in the repository can find repository-local setup guidance.
205
274
  Use `--dry-run` only when you want a preview without writing files or creating
206
275
  labels.
207
276
 
208
- ## Project Policy
277
+ ## What The Runner Checks
209
278
 
210
- Every installed repository owns its config:
279
+ Before a result becomes a draft PR, the runner checks the whole local result:
211
280
 
212
- ```sh
213
- .codex-orchestrator/config.json
214
- ```
281
+ - committed changes;
282
+ - staged changes;
283
+ - unstaged changes;
284
+ - untracked files;
285
+ - the completion report Codex was required to write;
286
+ - configured commands such as tests or type checks;
287
+ - review gates such as TDD evidence, changed tests, cleanup review, code review,
288
+ or acceptance proof when enabled;
289
+ - blocked paths and unsafe actions.
215
290
 
216
- That config is where the repo decides how strict automation should be. It
217
- controls the GitHub repo, labels, base branch, branch names, validation checks,
218
- review gates, blocked paths, child issue concurrency, durable logs, PR titles,
219
- and the prompts used for planning and implementation.
220
-
221
- The package ships bundled workflow prompts, so a repository does not need local
222
- Codex `SKILL.md` files installed on the user's machine. Setup copies those
223
- prompts into `.codex-orchestrator/prompts/workflows/`, and the runner reads the
224
- copied prompt files during `agent:auto` and `agent:plan-auto` runs. Workflow
225
- `skillName` values in config are descriptive metadata for the workflow role;
226
- they are not a runtime dependency on the user's local Codex skill directory.
227
- Setup also writes `.codex-orchestrator/prompts/manifest.json`, which lets later
228
- setup runs tell apart untouched package prompts from prompts edited by the
229
- project. By default, setup refreshes untouched prompts and reports conflicts for
230
- locally edited prompts.
231
-
232
- Configured checks run before publication. By default, missing
233
- `npm run <script>` checks are reported as skipped warnings, not failures. You can
234
- change that with `checksPolicy.missingNpmScript`.
235
-
236
- For repos with existing lint debt, `checksPolicy.lintBaseline.mode` can be set
237
- to `touched-only`. That lets a repo-wide lint failure be downgraded when a
238
- separate touched-files lint command passes.
239
-
240
- The default quality gate is conservative for runtime code changes. It can
241
- require TDD evidence, changed tests, code review, cleanup review for larger
242
- changes, and visual proof for UI work.
243
-
244
- ## Diagnostics
245
-
246
- `doctor` is a read-only readiness check for operators. It validates the target
247
- config, GitHub label visibility, git/base branch access, runner state paths,
248
- configured checks, the Codex command, phase profiles, and visual proof settings.
249
- It never launches Codex, creates worktrees, edits labels, or changes issues.
291
+ If the result passes, the runner pushes the branch and opens a draft PR. If it
292
+ does not pass, the runner marks the issue blocked, keeps the useful local
293
+ evidence, and explains what needs attention.
250
294
 
251
- ```sh
252
- codex-orchestrator doctor --target .
253
- codex-orchestrator doctor --target . --json
254
- ```
295
+ Acceptance proof is runner-owned. Codex can change product behavior, but the
296
+ runner runs proof afterwards and attaches screenshots, UI dumps, logs, smoke
297
+ outputs, or other artifacts to the PR and issue report. The proof phase must
298
+ produce a structured report that maps each required criterion to high-confidence
299
+ evidence. UI artifacts must include UI Evidence Contract mapping, and legacy
300
+ visual proof config only supplies migration inputs for report-producing proof.
255
301
 
256
- `status --json` returns the same queue view as text status plus active local
257
- runs and recent lifecycle events. The JSON is designed for wrappers and
258
- dashboards; it includes bounded artifact paths such as context snapshots, but
259
- not raw Codex transcripts, secrets, prompt text, or full issue comments.
260
-
261
- Codex command profiles can be set per runner phase under `codex.profiles`.
262
- Supported phases are `plan-parent`, `scoped-issue`, `tree-child`,
263
- `fresh-context-review`, `visual-proof`, and `quality-review`. Missing profile
264
- fields fall back to the global `codex.command`, `codex.args`, `timeoutMs`, and
265
- `idleTimeoutMs`, so existing configs keep working.
266
-
267
- Each Codex session writes a bounded context snapshot before invocation and links
268
- it from lifecycle events under the runner state directory. Snapshots record the
269
- issue identity, runner decision, selected profile, workspace paths, and
270
- publication boundaries so a maintainer can reproduce why a session started
271
- without reading raw logs.
272
-
273
- ## Visual Proof
274
-
275
- For browser UI work, configure a runner-owned proof command, usually a
276
- Playwright script:
277
-
278
- ```json
279
- {
280
- "reviewGates": {
281
- "visualProof": {
282
- "runnerValidationCommand": "npm run visual-proof -- --issue ${issueNumber}",
283
- "runnerTimeoutMs": 900000,
284
- "envPassthrough": [
285
- "CODEX_ORCHESTRATOR_LOGIN_EMAIL",
286
- "CODEX_ORCHESTRATOR_LOGIN_PASSWORD"
287
- ]
288
- }
289
- }
290
- }
291
- ```
302
+ ## Repository Policy
303
+
304
+ Every installed repository owns its automation policy in:
292
305
 
293
- The runner executes this command from the issue worktree after Codex finishes
294
- and before review-gate evaluation. It sets environment variables for the issue
295
- number, artifact directory, proof directory, Playwright profile directory,
296
- worktree path, and changed files.
306
+ ```sh
307
+ .codex-orchestrator/config.json
308
+ ```
297
309
 
298
- Screenshots created under `CODEX_ORCHESTRATOR_PROOF_DIR` are attached to the PR
299
- and issue review report. Keep login credentials outside config and expose only
300
- their variable names through `envPassthrough`.
310
+ That config controls labels, branch names, checks, review gates, blocked paths,
311
+ prompt files, child concurrency, and PR titles. The package provides defaults;
312
+ the target repository decides how strict they should be.
301
313
 
302
- For Android UI work, the implementation prompt asks Codex to use `adb` or an
303
- emulator-backed proof path instead of browser proof. Missing Android tooling or
304
- no usable device is reported as a warning with the concrete reason, not as an
305
- automatic release blocker. Native Android proof uses the project Gradle wrapper
306
- with a writable Gradle cache; Flutter-specific SDK cache recovery is used only
307
- for Flutter projects and only through a preconfigured writable SDK path in
308
- `CODEX_ORCHESTRATOR_FLUTTER_ROOT`. Native iOS proof uses Xcode simulator/device
309
- tooling with a writable DerivedData path.
314
+ For the full config surface and technical behavior, see
315
+ [docs/deep-dive.md](docs/deep-dive.md).
310
316
 
311
317
  ## Safety Model
312
318
 
313
319
  The package is PR-first and human-reviewed. The important guardrails are:
314
320
 
315
- - no automatic merge, and only draft PRs are opened;
321
+ - no automatic merge;
322
+ - draft PRs only;
316
323
  - Codex may change files, but the runner owns remote publication and GitHub
317
- state;
324
+ state changes;
318
325
  - only explicitly authorized issues run;
319
326
  - child issues are never inferred from ordinary links or references;
320
- - committed and uncommitted changes are checked before publication;
321
- - secret files, destructive data/cache actions, and production deploy/release
322
- actions are blocked by default;
323
- - malformed or missing completion reports block publication;
324
- - bounded rework stops at the configured limit;
325
- - Policy Suggestions are recommendations only;
326
- - underspecified work can be blocked for maintainer clarification instead of
327
- letting Codex invent product decisions.
327
+ - committed and uncommitted changes are checked;
328
+ - missing or malformed completion reports block publication;
329
+ - secret files, destructive data/cache actions, and production deploy or release
330
+ actions are blocked by default.
328
331
 
329
332
  ## Labels
330
333
 
@@ -352,47 +355,15 @@ codex-orchestrator setup [--target <path>] [--github-owner <owner>] \
352
355
  [--github-repo <repo>] [--dry-run] [--prepare-labels]
353
356
  codex-orchestrator status --target <path> [--dry-run] [--json]
354
357
  codex-orchestrator run --target <path> --issue <number>
358
+ codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
359
+ codex-orchestrator visual-proof android --issue <number> [--target <path>]
360
+ codex-orchestrator visual-proof ios --issue <number> [--target <path>]
355
361
  codex-orchestrator daemon --target <path> [--once] \
356
362
  [--interval-seconds <seconds>] [--max-runs <count>] \
357
363
  [--concurrency <count>]
358
364
  ```
359
365
 
360
- `setup` creates project-local config and prompt files under
361
- `.codex-orchestrator/`. Useful flags:
362
-
363
- - `--dry-run` - show the setup plan without writing files or creating labels;
364
- - `--prepare-labels` - create missing GitHub labels;
365
- - `--target <path>` - override the target directory, which defaults to the current directory;
366
- - `--github-owner <owner>` - override the GitHub owner inferred from `origin`;
367
- - `--github-repo <repo>` - override the GitHub repo inferred from `origin`;
368
- - `--sync-prompts <auto|keep|replace|merge>` - choose how package-bundled
369
- prompt updates are applied. `auto` refreshes untouched prompts and reports
370
- local-edit conflicts, `keep` preserves existing prompts, `replace` overwrites
371
- with bundled prompts, and `merge` appends bundled updates to locally edited
372
- prompts;
373
- - `--replace-package-skills` - refresh package-bundled prompt files. The flag
374
- name is kept for compatibility and behaves like `--sync-prompts=replace`; it
375
- does not install or require local Codex skills.
376
-
377
- Setup does not launch Codex, commit changes, or open pull requests.
378
- When the target repository already has a `package.json`, setup also adds
379
- `orchestrator:*` npm scripts. Daemon scripts run `doctor` first, then start the
380
- daemon only if the readiness check passes.
381
-
382
- `status` is read-only. It shows eligible issues, skipped issues with reasons,
383
- and local recovery state.
384
-
385
- `run` executes one selected issue when labels and state allow it. `agent:auto`
386
- opens one scoped draft PR. `agent:plan-auto` runs parent planning, child waves,
387
- final validation, and one integration draft PR.
388
-
389
- `daemon` polls for eligible work. By default, fresh setup config allows up to
390
- three scoped issues per batch through `runner.maxParallelScopedIssues`; legacy
391
- configs without that field remain sequential unless `--concurrency` is passed.
392
- Only scoped issues with non-overlapping ownership metadata can share a batch.
393
- `agent:plan-auto` runs remain exclusive. The daemon also cleans up runner-owned
394
- worktrees after their PRs are merged, while preserving dirty, blocked, active,
395
- or unpublished worktrees for inspection.
366
+ Use `codex-orchestrator <command> --help` for command-specific flags.
396
367
 
397
368
  ## Current Scope
398
369
 
package/dist/src/cli.js CHANGED
@@ -8,7 +8,11 @@ import { runDaemonCommand } from './runner/daemon-command.js';
8
8
  import { runDoctorCommand } from './runner/doctor-command.js';
9
9
  import { runPlanAutoCommand } from './runner/plan-auto-command.js';
10
10
  import { runScopedAutoCommand } from './runner/scoped-auto-command.js';
11
+ import { recoverScopedRun } from './runner/scoped-recovery.js';
11
12
  import { runStatusCommand } from './runner/status-command.js';
13
+ import { parseAndroidVisualProofArgs, runAndroidVisualProofCommand } from './runner/android-visual-proof-command.js';
14
+ import { parseIosVisualProofArgs, runIosVisualProofCommand } from './runner/ios-visual-proof-command.js';
15
+ import { parseMobileVisualProofArgs, runMobileVisualProofCommand } from './runner/mobile-visual-proof-command.js';
12
16
  import { runSetupCommand } from './setup/setup-command.js';
13
17
  import { promptSyncModes } from './setup/prompt-sync.js';
14
18
  const helpText = `codex-orchestrator
@@ -22,6 +26,9 @@ Usage:
22
26
  codex-orchestrator status --target <path> [--dry-run] [--json]
23
27
  codex-orchestrator run --target <path> --issue <number>
24
28
  codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>] [--concurrency <number>]
29
+ codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
30
+ codex-orchestrator visual-proof android --issue <number> [--target <path>]
31
+ codex-orchestrator visual-proof ios --issue <number> [--target <path>]
25
32
 
26
33
  Commands:
27
34
  health Run a no-op local health check.
@@ -30,6 +37,7 @@ Commands:
30
37
  status Show eligible/skipped issue work and local recovery state.
31
38
  run Execute one authorized issue: scoped agent:auto or full agent:plan-auto issue tree.
32
39
  daemon Poll GitHub Issues and execute eligible autonomous work until stopped.
40
+ visual-proof Run package-owned proof commands used by review gates.
33
41
 
34
42
  Options:
35
43
  --help, -h Show this help.
@@ -164,6 +172,48 @@ async function main(args) {
164
172
  return 1;
165
173
  }
166
174
  }
175
+ if (command === 'visual-proof') {
176
+ const [kind, ...rest] = args.slice(1);
177
+ if (kind !== 'mobile' && kind !== 'android' && kind !== 'ios') {
178
+ process.stderr.write('visual-proof requires a supported kind: mobile, android, or ios\nRun codex-orchestrator --help for usage.\n');
179
+ return 2;
180
+ }
181
+ try {
182
+ if (kind === 'mobile') {
183
+ const parsed = parseMobileVisualProofArgs(rest);
184
+ if (!parsed.ok) {
185
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
186
+ return 2;
187
+ }
188
+ await runMobileVisualProofCommand(parsed.value);
189
+ process.stdout.write(`mobile visual proof captured for issue #${parsed.value.issueNumber}\n`);
190
+ }
191
+ else if (kind === 'ios') {
192
+ const parsed = parseIosVisualProofArgs(rest);
193
+ if (!parsed.ok) {
194
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
195
+ return 2;
196
+ }
197
+ await runIosVisualProofCommand(parsed.value);
198
+ process.stdout.write(`ios visual proof captured for issue #${parsed.value.issueNumber}\n`);
199
+ }
200
+ else {
201
+ const parsed = parseAndroidVisualProofArgs(rest);
202
+ if (!parsed.ok) {
203
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
204
+ return 2;
205
+ }
206
+ await runAndroidVisualProofCommand(parsed.value);
207
+ process.stdout.write(`android visual proof captured for issue #${parsed.value.issueNumber}\n`);
208
+ }
209
+ return 0;
210
+ }
211
+ catch (error) {
212
+ const message = error instanceof Error ? error.message : 'visual proof failed';
213
+ process.stderr.write(`${message}\n`);
214
+ return 1;
215
+ }
216
+ }
167
217
  process.stderr.write(`Unknown command: ${command}\nRun codex-orchestrator --help for usage.\n`);
168
218
  return 1;
169
219
  }
@@ -177,6 +227,15 @@ async function runIssueCommand(targetRootInput, issueNumber) {
177
227
  }
178
228
  const decision = discoverIssueWork([issue], config)[0];
179
229
  if (!decision || decision.kind !== 'eligible') {
230
+ const recovered = await recoverScopedRun({
231
+ targetRoot,
232
+ issueNumber,
233
+ invocation: 'targeted',
234
+ issueAdapter,
235
+ });
236
+ if (recovered.status !== 'not-recoverable') {
237
+ return { reportComment: recovered.reportComment };
238
+ }
180
239
  const reason = decision?.kind === 'skipped' ? decision.reason : 'not eligible';
181
240
  throw new Error(`Issue #${issueNumber} is not eligible for autonomous work: ${reason}`);
182
241
  }