codex-orchestrator 0.1.30 → 0.1.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/README.md +152 -164
  3. package/dist/src/cli.js +68 -1
  4. package/dist/src/cli.js.map +1 -1
  5. package/dist/src/codex/command-adapter.d.ts +2 -1
  6. package/dist/src/codex/command-adapter.d.ts.map +1 -1
  7. package/dist/src/codex/command-adapter.js +13 -2
  8. package/dist/src/codex/command-adapter.js.map +1 -1
  9. package/dist/src/codex/mobile-device-guard.d.ts +7 -0
  10. package/dist/src/codex/mobile-device-guard.d.ts.map +1 -0
  11. package/dist/src/codex/mobile-device-guard.js +63 -0
  12. package/dist/src/codex/mobile-device-guard.js.map +1 -0
  13. package/dist/src/config/constants.d.ts +1 -1
  14. package/dist/src/config/constants.d.ts.map +1 -1
  15. package/dist/src/config/constants.js +1 -0
  16. package/dist/src/config/constants.js.map +1 -1
  17. package/dist/src/config/schema.d.ts +16 -2
  18. package/dist/src/config/schema.d.ts.map +1 -1
  19. package/dist/src/config/schema.js +42 -0
  20. package/dist/src/config/schema.js.map +1 -1
  21. package/dist/src/github/gh-pull-request-adapter.d.ts +1 -0
  22. package/dist/src/github/gh-pull-request-adapter.d.ts.map +1 -1
  23. package/dist/src/github/gh-pull-request-adapter.js +20 -0
  24. package/dist/src/github/gh-pull-request-adapter.js.map +1 -1
  25. package/dist/src/github/pull-requests.d.ts +2 -0
  26. package/dist/src/github/pull-requests.d.ts.map +1 -1
  27. package/dist/src/github/pull-requests.js +9 -0
  28. package/dist/src/github/pull-requests.js.map +1 -1
  29. package/dist/src/runner/acceptance-proof-runner.d.ts +54 -0
  30. package/dist/src/runner/acceptance-proof-runner.d.ts.map +1 -0
  31. package/dist/src/runner/acceptance-proof-runner.js +229 -0
  32. package/dist/src/runner/acceptance-proof-runner.js.map +1 -0
  33. package/dist/src/runner/acceptance-proof.d.ts +121 -0
  34. package/dist/src/runner/acceptance-proof.d.ts.map +1 -0
  35. package/dist/src/runner/acceptance-proof.js +405 -0
  36. package/dist/src/runner/acceptance-proof.js.map +1 -0
  37. package/dist/src/runner/android-visual-proof-command.d.ts +26 -0
  38. package/dist/src/runner/android-visual-proof-command.d.ts.map +1 -0
  39. package/dist/src/runner/android-visual-proof-command.js +598 -0
  40. package/dist/src/runner/android-visual-proof-command.js.map +1 -0
  41. package/dist/src/runner/command-utils.d.ts.map +1 -1
  42. package/dist/src/runner/command-utils.js +55 -6
  43. package/dist/src/runner/command-utils.js.map +1 -1
  44. package/dist/src/runner/completion-report.d.ts +3 -1
  45. package/dist/src/runner/completion-report.d.ts.map +1 -1
  46. package/dist/src/runner/completion-report.js +2 -1
  47. package/dist/src/runner/completion-report.js.map +1 -1
  48. package/dist/src/runner/daemon-command.d.ts +1 -0
  49. package/dist/src/runner/daemon-command.d.ts.map +1 -1
  50. package/dist/src/runner/daemon-command.js +139 -14
  51. package/dist/src/runner/daemon-command.js.map +1 -1
  52. package/dist/src/runner/doctor-command.js +3 -3
  53. package/dist/src/runner/doctor-command.js.map +1 -1
  54. package/dist/src/runner/durable-run-summary.d.ts +3 -0
  55. package/dist/src/runner/durable-run-summary.d.ts.map +1 -1
  56. package/dist/src/runner/durable-run-summary.js +2 -0
  57. package/dist/src/runner/durable-run-summary.js.map +1 -1
  58. package/dist/src/runner/handoff-evidence.d.ts +5 -0
  59. package/dist/src/runner/handoff-evidence.d.ts.map +1 -1
  60. package/dist/src/runner/handoff-evidence.js +22 -0
  61. package/dist/src/runner/handoff-evidence.js.map +1 -1
  62. package/dist/src/runner/ios-visual-proof-command.d.ts +25 -0
  63. package/dist/src/runner/ios-visual-proof-command.d.ts.map +1 -0
  64. package/dist/src/runner/ios-visual-proof-command.js +355 -0
  65. package/dist/src/runner/ios-visual-proof-command.js.map +1 -0
  66. package/dist/src/runner/lifecycle-events.d.ts +1 -1
  67. package/dist/src/runner/lifecycle-events.d.ts.map +1 -1
  68. package/dist/src/runner/local-execution-session.d.ts +20 -6
  69. package/dist/src/runner/local-execution-session.d.ts.map +1 -1
  70. package/dist/src/runner/local-execution-session.js +98 -11
  71. package/dist/src/runner/local-execution-session.js.map +1 -1
  72. package/dist/src/runner/local-state.d.ts +6 -0
  73. package/dist/src/runner/local-state.d.ts.map +1 -1
  74. package/dist/src/runner/local-state.js +14 -0
  75. package/dist/src/runner/local-state.js.map +1 -1
  76. package/dist/src/runner/mobile-device-lease.d.ts +11 -0
  77. package/dist/src/runner/mobile-device-lease.d.ts.map +1 -0
  78. package/dist/src/runner/mobile-device-lease.js +121 -0
  79. package/dist/src/runner/mobile-device-lease.js.map +1 -0
  80. package/dist/src/runner/mobile-visual-proof-command.d.ts +17 -0
  81. package/dist/src/runner/mobile-visual-proof-command.d.ts.map +1 -0
  82. package/dist/src/runner/mobile-visual-proof-command.js +168 -0
  83. package/dist/src/runner/mobile-visual-proof-command.js.map +1 -0
  84. package/dist/src/runner/plan-auto-command.d.ts.map +1 -1
  85. package/dist/src/runner/plan-auto-command.js +58 -7
  86. package/dist/src/runner/plan-auto-command.js.map +1 -1
  87. package/dist/src/runner/prompt.js +2 -2
  88. package/dist/src/runner/prompt.js.map +1 -1
  89. package/dist/src/runner/recovery.d.ts +2 -1
  90. package/dist/src/runner/recovery.d.ts.map +1 -1
  91. package/dist/src/runner/recovery.js +24 -0
  92. package/dist/src/runner/recovery.js.map +1 -1
  93. package/dist/src/runner/review-gate-policy.d.ts +7 -0
  94. package/dist/src/runner/review-gate-policy.d.ts.map +1 -1
  95. package/dist/src/runner/review-gate-policy.js +88 -22
  96. package/dist/src/runner/review-gate-policy.js.map +1 -1
  97. package/dist/src/runner/review-gates.d.ts.map +1 -1
  98. package/dist/src/runner/review-gates.js +44 -11
  99. package/dist/src/runner/review-gates.js.map +1 -1
  100. package/dist/src/runner/rework-policy.d.ts +1 -0
  101. package/dist/src/runner/rework-policy.d.ts.map +1 -1
  102. package/dist/src/runner/rework-policy.js +7 -0
  103. package/dist/src/runner/rework-policy.js.map +1 -1
  104. package/dist/src/runner/scoped-auto-command.d.ts +46 -1
  105. package/dist/src/runner/scoped-auto-command.d.ts.map +1 -1
  106. package/dist/src/runner/scoped-auto-command.js +148 -48
  107. package/dist/src/runner/scoped-auto-command.js.map +1 -1
  108. package/dist/src/runner/scoped-recovery.d.ts +65 -0
  109. package/dist/src/runner/scoped-recovery.d.ts.map +1 -0
  110. package/dist/src/runner/scoped-recovery.js +484 -0
  111. package/dist/src/runner/scoped-recovery.js.map +1 -0
  112. package/dist/src/runner/status-command.d.ts.map +1 -1
  113. package/dist/src/runner/status-command.js +1 -0
  114. package/dist/src/runner/status-command.js.map +1 -1
  115. package/dist/src/runner/visual-proof-runner.d.ts +1 -0
  116. package/dist/src/runner/visual-proof-runner.d.ts.map +1 -1
  117. package/dist/src/runner/visual-proof-runner.js +144 -16
  118. package/dist/src/runner/visual-proof-runner.js.map +1 -1
  119. package/dist/src/setup/project-config.d.ts +4 -0
  120. package/dist/src/setup/project-config.d.ts.map +1 -1
  121. package/dist/src/setup/project-config.js +142 -9
  122. package/dist/src/setup/project-config.js.map +1 -1
  123. package/dist/src/setup/setup-command.d.ts.map +1 -1
  124. package/dist/src/setup/setup-command.js +2 -2
  125. package/dist/src/setup/setup-command.js.map +1 -1
  126. package/dist/src/setup/workflows.d.ts.map +1 -1
  127. package/dist/src/setup/workflows.js +5 -0
  128. package/dist/src/setup/workflows.js.map +1 -1
  129. package/docs/deep-dive.md +230 -31
  130. package/package.json +1 -1
  131. package/prompts/workflows/acceptance-proof.md +25 -0
  132. package/prompts/workflows/issue-breakdown.md +7 -0
  133. package/prompts/workflows/scoped-implementation.md +6 -2
package/CHANGELOG.md CHANGED
@@ -6,6 +6,71 @@ The format is based on Keep a Changelog, and this project follows SemVer.
6
6
 
7
7
  ## [Unreleased]
8
8
 
9
+ ## [0.1.35] - 2026-05-21
10
+
11
+ ### Added
12
+ - Added a runner-validated UI Evidence Contract for Acceptance Proof reports,
13
+ covering workflow, viewport, freshness, layout, copy, and source-input
14
+ evidence for screenshot and UI-dump artifacts.
15
+ - Added live smoke coverage for UI Evidence pass and blocking cases, including
16
+ missing UI Evidence and too-narrow desktop viewport proof.
17
+
18
+ ### Changed
19
+ - Runner-owned visual proof no longer treats screenshot-only command success as
20
+ a pass path; proof commands must produce a valid machine-readable Acceptance
21
+ Proof report.
22
+ - Updated the legacy `visual-proof` live smoke scenario to emit the same
23
+ machine-readable UI Evidence report required by the runner.
24
+
25
+ ## [0.1.34] - 2026-05-20
26
+
27
+ ### Added
28
+ - Added an Adaptive Proof Agent Codex phase for scoped and issue-tree child
29
+ runs, with runner-provided proof report paths, artifact directories, changed
30
+ file context, and proof-owned repair policy.
31
+ - Added durable Acceptance Proof attempt evidence in lifecycle events, run
32
+ summaries, blocked comments, review reports, and issue-tree PR handoff.
33
+ - Added package-bundled Acceptance Proof workflow prompts and setup routing for
34
+ the new proof phase.
35
+
36
+ ### Changed
37
+ - Parent `agent:plan-auto` child waves now block parent publication when a child
38
+ Acceptance Proof attempt fails, requests rework, or is blocked.
39
+ - Proof attempts now use isolated Codex homes and preserve proof artifacts while
40
+ keeping publication authority runner-owned.
41
+
42
+ ## [0.1.33] - 2026-05-20
43
+
44
+ ### Added
45
+ - Added canonical `reviewGates.acceptanceProof` policy with proof-owned path
46
+ classification and machine validation for high-confidence proof reports.
47
+ - Added live smoke scenarios for canonical Acceptance Proof pass, proof rework,
48
+ low-confidence blocking, and proof-phase product-diff blocking.
49
+
50
+ ### Changed
51
+ - Kept `reviewGates.visualProof` as a compatibility adapter while routing
52
+ runner prompts and proof policy through Acceptance Proof.
53
+ - Proof-phase product-code changes now block publishability instead of being
54
+ silently committed as verification output.
55
+
56
+ ## [0.1.32] - 2026-05-19
57
+
58
+ ### Added
59
+ - Added package-owned `visual-proof mobile`, `visual-proof android`, and
60
+ `visual-proof ios` commands for reusable UI launch proof across installed
61
+ repositories.
62
+ - Mobile visual proof now supports Flutter Android, native Android, Flutter iOS,
63
+ and native iOS projects, with screenshots saved into runner proof artifacts.
64
+
65
+ ### Changed
66
+ - Setup now defaults visual proof to
67
+ `codex-orchestrator visual-proof mobile --issue ${issueNumber}` instead of a
68
+ target-repo local proof script.
69
+ - Android proof resolves SDK tools from environment variables, `PATH`, and
70
+ default macOS, Linux, and Windows SDK locations.
71
+ - On macOS, mobile proof falls back to the iOS simulator when Android tooling or
72
+ devices are unavailable and the repo has an iOS target.
73
+
9
74
  ## [0.1.30] - 2026-05-18
10
75
 
11
76
  ### Added
package/README.md CHANGED
@@ -1,6 +1,10 @@
1
1
  # codex-orchestrator
2
2
 
3
- `codex-orchestrator` turns GitHub Issues into controlled Codex work.
3
+ ## About
4
+
5
+ `codex-orchestrator` turns GitHub Issues and project work into isolated,
6
+ autonomous Codex implementation runs, allowing maintainers to manage work
7
+ instead of supervising coding agents.
4
8
 
5
9
  Instead of starting a new Codex chat for every issue, you label the work you
6
10
  want automated. The runner creates an isolated workspace, gives Codex the issue
@@ -41,23 +45,76 @@ control to humans before anything is merged.
41
45
  - A repeatable way to send selected GitHub Issues to Codex.
42
46
  - One-off autonomous runs for scoped implementation tasks.
43
47
  - Parent planning for larger features, with child issues executed in safe waves.
44
- - Project-owned rules for labels, branches, prompts, checks, review gates, and
45
- blocked actions.
46
- - Full change-set checks, including local commits, staged files, unstaged files,
47
- and untracked files.
48
- - Durable logs and summaries when a run is interrupted, blocked, or ready for
49
- review.
48
+ - Project-owned rules for what Codex may run, how results are checked, and when
49
+ a human must step in.
50
+ - Adaptive Acceptance Proof for runner-owned verification of UI, API, worker,
51
+ CLI, browser, mobile, and live-smoke behavior before draft PR handoff.
52
+ - Logs, summaries, and proof artifacts when available.
53
+ - Recovery for interrupted runner handoff when Codex finished locally but the
54
+ draft PR was not created yet.
50
55
  - Draft PR handoff by default. No auto-merge.
51
56
 
52
57
  ## How It Works
53
58
 
54
- There are two main modes.
59
+ At a high level, GitHub Issues are the queue, labels authorize work, isolated
60
+ worktrees keep runs separate, and the runner owns validation and publication.
61
+
62
+ ```mermaid
63
+ flowchart TD
64
+ A["Target repo"] --> B["codex-orchestrator setup"]
65
+ B --> C[".codex-orchestrator/config.json + prompts"]
66
+ C --> D["GitHub Issue gets agent:auto or agent:plan-auto"]
67
+ D --> E["status / daemon / run"]
68
+ E --> R{"Recover interrupted handoff?"}
69
+ R -- "yes" --> L
70
+ R -- "no" --> F{"Eligible?"}
71
+ F -- "no" --> G["Skipped with reason"]
72
+ F -- "yes" --> H["Runner claims issue: agent:running"]
73
+ H --> I["Create isolated branch + worktree"]
74
+ I --> J["Build Codex prompt from issue + repo policy"]
75
+ J --> K["Run Codex CLI"]
76
+ K --> AAP{"Adaptive Acceptance Proof required?"}
77
+ AAP -- "no" --> L["Runner validates the full changeset"]
78
+ AAP -- "yes" --> AP["Run proof phase and collect artifacts"]
79
+ AP --> APR{"Proof result"}
80
+ APR -- "passed" --> L
81
+ APR -- "needs rework" --> J
82
+ APR -- "blocked" --> N
83
+ L --> M{"Gates pass?"}
84
+ M -- "no" --> N["Mark blocked, preserve evidence"]
85
+ M -- "yes" --> O["Push branch"]
86
+ O --> P["Open draft PR"]
87
+ P --> Q["Move issue to agent:review + post report"]
88
+ ```
89
+
90
+ The important boundary is simple: Codex writes code, but the runner decides
91
+ whether that code can be handed to humans. The runner owns checks, acceptance
92
+ proof, labels, comments, branch pushes, and draft PR creation.
93
+
94
+ ### Adaptive Acceptance Proof
95
+
96
+ Adaptive Acceptance Proof is the runner-owned verification phase for work that
97
+ needs observable product proof. After implementation, the runner can start a
98
+ separate proof phase that inspects the issue, changed files, and acceptance
99
+ criteria; runs focused browser, mobile, API, worker, CLI, or live-smoke checks;
100
+ and writes a machine-readable proof report with artifact links.
101
+
102
+ A result can reach draft PR handoff only when every required criterion maps to
103
+ high-confidence artifact evidence. If proof finds missing behavior, it returns a
104
+ concrete rework request and the runner loops back through implementation within
105
+ the configured iteration limit. If proof is malformed, low-confidence, lacks
106
+ artifacts, or changes product code during verification, the runner blocks
107
+ publication and preserves the evidence.
108
+
109
+ For UI proof, screenshots and UI dumps must also satisfy the UI Evidence
110
+ Contract: exact workflow, viewport coverage, current artifact freshness, layout
111
+ review, copy review, and source inputs. Screenshot-only proof cannot pass.
112
+
113
+ There are two main ways to run work.
55
114
 
56
115
  ### `agent:auto`
57
116
 
58
- Use `agent:auto` for one clear standalone implementation issue. Do not use it
59
- for child issues created by `agent:plan-auto`; those are marked with
60
- `agent:child` and are executed only by the parent issue-tree flow.
117
+ Use `agent:auto` for one clear standalone implementation issue.
61
118
 
62
119
  The runner:
63
120
 
@@ -69,18 +126,26 @@ The runner:
69
126
  6. Pushes the branch and opens a draft PR only after the gates pass.
70
127
  7. Moves the issue to review and posts the run report.
71
128
 
129
+ When the daemon runs more than one `agent:auto` issue at once, it only batches
130
+ issues whose declared ownership does not overlap. Issues without ownership
131
+ metadata still run, but conservatively.
132
+
72
133
  ### `agent:plan-auto`
73
134
 
74
- Use `agent:plan-auto` for work that needs planning first.
135
+ Use `agent:plan-auto` for larger work that should be planned before
136
+ implementation.
75
137
 
76
138
  The runner asks Codex to plan the parent issue, break it into child issues,
77
- triage them, run safe children in dependency order, and then open one
78
- integration draft PR.
139
+ run safe children in dependency order, merge successful child branches into one
140
+ integration branch, validate that integration branch, and then open one draft
141
+ PR.
142
+
143
+ Child issues created by this flow use `agent:child`, not `agent:auto`. They are
144
+ owned by the parent run and are not picked up as standalone daemon work.
79
145
 
80
- Only child issues explicitly marked by the runner belong to the autonomous tree.
81
- Ordinary links, milestones, project fields, or casual references are not enough.
82
- Child issues use `agent:child`, not `agent:auto`, so the daemon cannot confuse
83
- parent-owned child work with standalone scoped work.
146
+ If a runner stops after Codex finished locally but before draft PR handoff, the
147
+ runner can recover from its local state and completed report without rerunning
148
+ Codex. See [docs/deep-dive.md](docs/deep-dive.md) for the recovery rules.
84
149
 
85
150
  ## Basic Workflow
86
151
 
@@ -94,7 +159,13 @@ parent-owned child work with standalone scoped work.
94
159
 
95
160
  The runner never auto-merges.
96
161
 
97
- ## Installation
162
+ ## Agent Memory
163
+
164
+ Repo-local Dreaming-lite memory lives in `docs/agents/memory/`. It is a small
165
+ curated lessons cache for repeated runner/debug/agent-workflow patterns, not a
166
+ replacement for `AGENTS.md`, ADRs, `docs/deep-dive.md`, or package prompts.
167
+
168
+ ## Install
98
169
 
99
170
  Requirements:
100
171
 
@@ -123,7 +194,7 @@ You can also run it with `npx`:
123
194
  npx codex-orchestrator --help
124
195
  ```
125
196
 
126
- ## Quick Start
197
+ ## Set Up A Repository
127
198
 
128
199
  Open the repository that should receive autonomous Codex work:
129
200
 
@@ -137,19 +208,19 @@ Run setup and create missing labels:
137
208
  codex-orchestrator setup --prepare-labels
138
209
  ```
139
210
 
140
- By default, setup reads the GitHub owner and repository name from `git remote
141
- origin` and uses the current directory as the target repository. Use `--target`,
142
- `--github-owner`, and `--github-repo` only when you need to override those
143
- defaults.
144
-
145
211
  Commit the generated `.codex-orchestrator/` directory to your repository. It is
146
212
  the repository-owned policy for how autonomous work should run.
147
213
 
148
- Check eligible work:
214
+ By default, setup reads the GitHub owner and repo from `git remote origin`. Use
215
+ `--target`, `--github-owner`, or `--github-repo` only when you need to override
216
+ that.
217
+
218
+ ## Run Work
219
+
220
+ Check what the runner can see:
149
221
 
150
222
  ```sh
151
223
  codex-orchestrator status --target .
152
- codex-orchestrator status --target . --json
153
224
  codex-orchestrator doctor --target .
154
225
  ```
155
226
 
@@ -165,6 +236,16 @@ Run the daemon:
165
236
  codex-orchestrator daemon --target .
166
237
  ```
167
238
 
239
+ Run up to three independent scoped issues at once:
240
+
241
+ ```sh
242
+ codex-orchestrator daemon --target . --concurrency 3
243
+ ```
244
+
245
+ `status` and `doctor` are read-only. `run` executes one selected issue.
246
+ `daemon` polls for eligible work and starts safe runs according to the policy in
247
+ `.codex-orchestrator/config.json`.
248
+
168
249
  ## Agent-Assisted Setup
169
250
 
170
251
  You do not need a long prompt. You can ask an agent:
@@ -193,126 +274,60 @@ working in the repository can find repository-local setup guidance.
193
274
  Use `--dry-run` only when you want a preview without writing files or creating
194
275
  labels.
195
276
 
196
- ## Project Policy
277
+ ## What The Runner Checks
197
278
 
198
- Every installed repository owns its config:
279
+ Before a result becomes a draft PR, the runner checks the whole local result:
199
280
 
200
- ```sh
201
- .codex-orchestrator/config.json
202
- ```
281
+ - committed changes;
282
+ - staged changes;
283
+ - unstaged changes;
284
+ - untracked files;
285
+ - the completion report Codex was required to write;
286
+ - configured commands such as tests or type checks;
287
+ - review gates such as TDD evidence, changed tests, cleanup review, code review,
288
+ or acceptance proof when enabled;
289
+ - blocked paths and unsafe actions.
203
290
 
204
- That config is where the repo decides how strict automation should be. It
205
- controls the GitHub repo, labels, base branch, branch names, validation checks,
206
- review gates, blocked paths, child issue concurrency, durable logs, PR titles,
207
- and the prompts used for planning and implementation.
208
-
209
- The package ships bundled workflow prompts, so a repository does not need local
210
- Codex `SKILL.md` files installed on the user's machine. Setup copies those
211
- prompts into `.codex-orchestrator/prompts/workflows/`, and the runner reads the
212
- copied prompt files during `agent:auto` and `agent:plan-auto` runs. Workflow
213
- `skillName` values in config are descriptive metadata for the workflow role;
214
- they are not a runtime dependency on the user's local Codex skill directory.
215
- Setup also writes `.codex-orchestrator/prompts/manifest.json`, which lets later
216
- setup runs tell apart untouched package prompts from prompts edited by the
217
- project. By default, setup refreshes untouched prompts and reports conflicts for
218
- locally edited prompts.
219
-
220
- Configured checks run before publication. By default, missing
221
- `npm run <script>` checks are reported as skipped warnings, not failures. You can
222
- change that with `checksPolicy.missingNpmScript`.
223
-
224
- For repos with existing lint debt, `checksPolicy.lintBaseline.mode` can be set
225
- to `touched-only`. That lets a repo-wide lint failure be downgraded when a
226
- separate touched-files lint command passes.
227
-
228
- The default quality gate is conservative for runtime code changes. It can
229
- require TDD evidence, changed tests, code review, cleanup review for larger
230
- changes, and visual proof for UI work.
231
-
232
- ## Diagnostics
233
-
234
- `doctor` is a read-only readiness check for operators. It validates the target
235
- config, GitHub label visibility, git/base branch access, runner state paths,
236
- configured checks, the Codex command, phase profiles, and visual proof settings.
237
- It never launches Codex, creates worktrees, edits labels, or changes issues.
291
+ If the result passes, the runner pushes the branch and opens a draft PR. If it
292
+ does not pass, the runner marks the issue blocked, keeps the useful local
293
+ evidence, and explains what needs attention.
238
294
 
239
- ```sh
240
- codex-orchestrator doctor --target .
241
- codex-orchestrator doctor --target . --json
242
- ```
295
+ Acceptance proof is runner-owned. Codex can change product behavior, but the
296
+ runner runs proof afterwards and attaches screenshots, UI dumps, logs, smoke
297
+ outputs, or other artifacts to the PR and issue report. The proof phase must
298
+ produce a structured report that maps each required criterion to high-confidence
299
+ evidence. UI artifacts must include UI Evidence Contract mapping, and legacy
300
+ visual proof config only supplies migration inputs for report-producing proof.
243
301
 
244
- `status --json` returns the same queue view as text status plus active local
245
- runs and recent lifecycle events. The JSON is designed for wrappers and
246
- dashboards; it includes bounded artifact paths such as context snapshots, but
247
- not raw Codex transcripts, secrets, prompt text, or full issue comments.
248
-
249
- Codex command profiles can be set per runner phase under `codex.profiles`.
250
- Supported phases are `plan-parent`, `scoped-issue`, `tree-child`,
251
- `fresh-context-review`, `visual-proof`, and `quality-review`. Missing profile
252
- fields fall back to the global `codex.command`, `codex.args`, `timeoutMs`, and
253
- `idleTimeoutMs`, so existing configs keep working.
254
-
255
- Each Codex session writes a bounded context snapshot before invocation and links
256
- it from lifecycle events under the runner state directory. Snapshots record the
257
- issue identity, runner decision, selected profile, workspace paths, and
258
- publication boundaries so a maintainer can reproduce why a session started
259
- without reading raw logs.
260
-
261
- ## Visual Proof
262
-
263
- For browser UI work, configure a runner-owned proof command, usually a
264
- Playwright script:
265
-
266
- ```json
267
- {
268
- "reviewGates": {
269
- "visualProof": {
270
- "runnerValidationCommand": "npm run visual-proof -- --issue ${issueNumber}",
271
- "runnerTimeoutMs": 900000,
272
- "envPassthrough": [
273
- "CODEX_ORCHESTRATOR_LOGIN_EMAIL",
274
- "CODEX_ORCHESTRATOR_LOGIN_PASSWORD"
275
- ]
276
- }
277
- }
278
- }
279
- ```
302
+ ## Repository Policy
303
+
304
+ Every installed repository owns its automation policy in:
280
305
 
281
- The runner executes this command from the issue worktree after Codex finishes
282
- and before review-gate evaluation. It sets environment variables for the issue
283
- number, artifact directory, proof directory, Playwright profile directory,
284
- worktree path, and changed files.
306
+ ```sh
307
+ .codex-orchestrator/config.json
308
+ ```
285
309
 
286
- Screenshots created under `CODEX_ORCHESTRATOR_PROOF_DIR` are attached to the PR
287
- and issue review report. Keep login credentials outside config and expose only
288
- their variable names through `envPassthrough`.
310
+ That config controls labels, branch names, checks, review gates, blocked paths,
311
+ prompt files, child concurrency, and PR titles. The package provides defaults;
312
+ the target repository decides how strict they should be.
289
313
 
290
- For Android UI work, the implementation prompt asks Codex to use `adb` or an
291
- emulator-backed proof path instead of browser proof. Missing Android tooling or
292
- no usable device is reported as a warning with the concrete reason, not as an
293
- automatic release blocker. Native Android proof uses the project Gradle wrapper
294
- with a writable Gradle cache; Flutter-specific SDK cache recovery is used only
295
- for Flutter projects and only through a preconfigured writable SDK path in
296
- `CODEX_ORCHESTRATOR_FLUTTER_ROOT`. Native iOS proof uses Xcode simulator/device
297
- tooling with a writable DerivedData path.
314
+ For the full config surface and technical behavior, see
315
+ [docs/deep-dive.md](docs/deep-dive.md).
298
316
 
299
317
  ## Safety Model
300
318
 
301
319
  The package is PR-first and human-reviewed. The important guardrails are:
302
320
 
303
- - no automatic merge, and only draft PRs are opened;
321
+ - no automatic merge;
322
+ - draft PRs only;
304
323
  - Codex may change files, but the runner owns remote publication and GitHub
305
- state;
324
+ state changes;
306
325
  - only explicitly authorized issues run;
307
326
  - child issues are never inferred from ordinary links or references;
308
- - committed and uncommitted changes are checked before publication;
309
- - secret files, destructive data/cache actions, and production deploy/release
310
- actions are blocked by default;
311
- - malformed or missing completion reports block publication;
312
- - bounded rework stops at the configured limit;
313
- - Policy Suggestions are recommendations only;
314
- - underspecified work can be blocked for maintainer clarification instead of
315
- letting Codex invent product decisions.
327
+ - committed and uncommitted changes are checked;
328
+ - missing or malformed completion reports block publication;
329
+ - secret files, destructive data/cache actions, and production deploy or release
330
+ actions are blocked by default.
316
331
 
317
332
  ## Labels
318
333
 
@@ -340,42 +355,15 @@ codex-orchestrator setup [--target <path>] [--github-owner <owner>] \
340
355
  [--github-repo <repo>] [--dry-run] [--prepare-labels]
341
356
  codex-orchestrator status --target <path> [--dry-run] [--json]
342
357
  codex-orchestrator run --target <path> --issue <number>
358
+ codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
359
+ codex-orchestrator visual-proof android --issue <number> [--target <path>]
360
+ codex-orchestrator visual-proof ios --issue <number> [--target <path>]
343
361
  codex-orchestrator daemon --target <path> [--once] \
344
- [--interval-seconds <seconds>] [--max-runs <count>]
362
+ [--interval-seconds <seconds>] [--max-runs <count>] \
363
+ [--concurrency <count>]
345
364
  ```
346
365
 
347
- `setup` creates project-local config and prompt files under
348
- `.codex-orchestrator/`. Useful flags:
349
-
350
- - `--dry-run` - show the setup plan without writing files or creating labels;
351
- - `--prepare-labels` - create missing GitHub labels;
352
- - `--target <path>` - override the target directory, which defaults to the current directory;
353
- - `--github-owner <owner>` - override the GitHub owner inferred from `origin`;
354
- - `--github-repo <repo>` - override the GitHub repo inferred from `origin`;
355
- - `--sync-prompts <auto|keep|replace|merge>` - choose how package-bundled
356
- prompt updates are applied. `auto` refreshes untouched prompts and reports
357
- local-edit conflicts, `keep` preserves existing prompts, `replace` overwrites
358
- with bundled prompts, and `merge` appends bundled updates to locally edited
359
- prompts;
360
- - `--replace-package-skills` - refresh package-bundled prompt files. The flag
361
- name is kept for compatibility and behaves like `--sync-prompts=replace`; it
362
- does not install or require local Codex skills.
363
-
364
- Setup does not launch Codex, commit changes, or open pull requests.
365
- When the target repository already has a `package.json`, setup also adds
366
- `orchestrator:*` npm scripts. Daemon scripts run `doctor` first, then start the
367
- daemon only if the readiness check passes.
368
-
369
- `status` is read-only. It shows eligible issues, skipped issues with reasons,
370
- and local recovery state.
371
-
372
- `run` executes one selected issue when labels and state allow it. `agent:auto`
373
- opens one scoped draft PR. `agent:plan-auto` runs parent planning, child waves,
374
- final validation, and one integration draft PR.
375
-
376
- `daemon` polls for eligible work and runs one issue at a time. It also cleans up
377
- runner-owned worktrees after their PRs are merged, while preserving dirty,
378
- blocked, active, or unpublished worktrees for inspection.
366
+ Use `codex-orchestrator <command> --help` for command-specific flags.
379
367
 
380
368
  ## Current Scope
381
369
 
package/dist/src/cli.js CHANGED
@@ -8,7 +8,11 @@ import { runDaemonCommand } from './runner/daemon-command.js';
8
8
  import { runDoctorCommand } from './runner/doctor-command.js';
9
9
  import { runPlanAutoCommand } from './runner/plan-auto-command.js';
10
10
  import { runScopedAutoCommand } from './runner/scoped-auto-command.js';
11
+ import { recoverScopedRun } from './runner/scoped-recovery.js';
11
12
  import { runStatusCommand } from './runner/status-command.js';
13
+ import { parseAndroidVisualProofArgs, runAndroidVisualProofCommand } from './runner/android-visual-proof-command.js';
14
+ import { parseIosVisualProofArgs, runIosVisualProofCommand } from './runner/ios-visual-proof-command.js';
15
+ import { parseMobileVisualProofArgs, runMobileVisualProofCommand } from './runner/mobile-visual-proof-command.js';
12
16
  import { runSetupCommand } from './setup/setup-command.js';
13
17
  import { promptSyncModes } from './setup/prompt-sync.js';
14
18
  const helpText = `codex-orchestrator
@@ -21,7 +25,10 @@ Usage:
21
25
  codex-orchestrator setup [--target <path>] [--github-owner <owner>] [--github-repo <repo>] [--dry-run] [--prepare-labels] [--sync-prompts <mode>]
22
26
  codex-orchestrator status --target <path> [--dry-run] [--json]
23
27
  codex-orchestrator run --target <path> --issue <number>
24
- codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>]
28
+ codex-orchestrator daemon --target <path> [--interval-seconds <number>] [--once] [--max-runs <number>] [--concurrency <number>]
29
+ codex-orchestrator visual-proof mobile --issue <number> [--target <path>]
30
+ codex-orchestrator visual-proof android --issue <number> [--target <path>]
31
+ codex-orchestrator visual-proof ios --issue <number> [--target <path>]
25
32
 
26
33
  Commands:
27
34
  health Run a no-op local health check.
@@ -30,6 +37,7 @@ Commands:
30
37
  status Show eligible/skipped issue work and local recovery state.
31
38
  run Execute one authorized issue: scoped agent:auto or full agent:plan-auto issue tree.
32
39
  daemon Poll GitHub Issues and execute eligible autonomous work until stopped.
40
+ visual-proof Run package-owned proof commands used by review gates.
33
41
 
34
42
  Options:
35
43
  --help, -h Show this help.
@@ -151,6 +159,7 @@ async function main(args) {
151
159
  intervalMs: parsed.value.intervalSeconds * 1000,
152
160
  once: parsed.value.once,
153
161
  maxRuns: parsed.value.maxRuns,
162
+ concurrency: parsed.value.concurrency,
154
163
  onEvent: (line) => {
155
164
  process.stdout.write(`${line}\n`);
156
165
  },
@@ -163,6 +172,48 @@ async function main(args) {
163
172
  return 1;
164
173
  }
165
174
  }
175
+ if (command === 'visual-proof') {
176
+ const [kind, ...rest] = args.slice(1);
177
+ if (kind !== 'mobile' && kind !== 'android' && kind !== 'ios') {
178
+ process.stderr.write('visual-proof requires a supported kind: mobile, android, or ios\nRun codex-orchestrator --help for usage.\n');
179
+ return 2;
180
+ }
181
+ try {
182
+ if (kind === 'mobile') {
183
+ const parsed = parseMobileVisualProofArgs(rest);
184
+ if (!parsed.ok) {
185
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
186
+ return 2;
187
+ }
188
+ await runMobileVisualProofCommand(parsed.value);
189
+ process.stdout.write(`mobile visual proof captured for issue #${parsed.value.issueNumber}\n`);
190
+ }
191
+ else if (kind === 'ios') {
192
+ const parsed = parseIosVisualProofArgs(rest);
193
+ if (!parsed.ok) {
194
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
195
+ return 2;
196
+ }
197
+ await runIosVisualProofCommand(parsed.value);
198
+ process.stdout.write(`ios visual proof captured for issue #${parsed.value.issueNumber}\n`);
199
+ }
200
+ else {
201
+ const parsed = parseAndroidVisualProofArgs(rest);
202
+ if (!parsed.ok) {
203
+ process.stderr.write(`${parsed.error}\nRun codex-orchestrator --help for usage.\n`);
204
+ return 2;
205
+ }
206
+ await runAndroidVisualProofCommand(parsed.value);
207
+ process.stdout.write(`android visual proof captured for issue #${parsed.value.issueNumber}\n`);
208
+ }
209
+ return 0;
210
+ }
211
+ catch (error) {
212
+ const message = error instanceof Error ? error.message : 'visual proof failed';
213
+ process.stderr.write(`${message}\n`);
214
+ return 1;
215
+ }
216
+ }
166
217
  process.stderr.write(`Unknown command: ${command}\nRun codex-orchestrator --help for usage.\n`);
167
218
  return 1;
168
219
  }
@@ -176,6 +227,15 @@ async function runIssueCommand(targetRootInput, issueNumber) {
176
227
  }
177
228
  const decision = discoverIssueWork([issue], config)[0];
178
229
  if (!decision || decision.kind !== 'eligible') {
230
+ const recovered = await recoverScopedRun({
231
+ targetRoot,
232
+ issueNumber,
233
+ invocation: 'targeted',
234
+ issueAdapter,
235
+ });
236
+ if (recovered.status !== 'not-recoverable') {
237
+ return { reportComment: recovered.reportComment };
238
+ }
179
239
  const reason = decision?.kind === 'skipped' ? decision.reason : 'not eligible';
180
240
  throw new Error(`Issue #${issueNumber} is not eligible for autonomous work: ${reason}`);
181
241
  }
@@ -307,6 +367,13 @@ function parseDaemonArgs(args) {
307
367
  parsed.maxRuns = Number(next);
308
368
  index += 1;
309
369
  break;
370
+ case '--concurrency':
371
+ if (!next || next.startsWith('--') || !Number.isInteger(Number(next)) || Number(next) < 1 || Number(next) > 3) {
372
+ return { ok: false, error: 'daemon requires --concurrency <integer between 1 and 3>' };
373
+ }
374
+ parsed.concurrency = Number(next);
375
+ index += 1;
376
+ break;
310
377
  default:
311
378
  return { ok: false, error: `Unknown daemon option: ${arg ?? ''}` };
312
379
  }