headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,3571 @@
1
+ /**
2
+ * `headlesscode orchestrate` subcommand — the thin orchestrator entry point.
3
+ *
4
+ * npx tsx src/cli.ts orchestrate --repo <path> --issue 27 --issue 29 \
5
+ * [--mode code] [--issues-json <path>] [--dry-run] [--no-review]
6
+ *
7
+ * Flow:
8
+ * 1. Read issues (`gh issue view --json number,title,body` when gh is
9
+ * available, else `--issues-json <file>`).
10
+ * 2. Split them into worktree groups with splitIssues (the
11
+ * multi-agent-orchestrator heuristics).
12
+ * 3. Generate task files under <repo>/plans/parallel-tasks/.
13
+ * 4. Spawn via scripts/spawn-parallel-worktrees.sh (the bash script remains
14
+ * the ACTUAL spawner — the CLI only assembles the spec triples).
15
+ * 5. Watch for completion (.harness.done markers) and run the headless
16
+ * reviewer on each group that finished cleanly.
17
+ *
18
+ * `--dry-run` prints the split plan + the exact spawn command without
19
+ * spawning anything (no API key needed).
20
+ */
21
+
22
+ import { execFileSync, spawn, spawnSync } from "node:child_process"
23
+ import * as fs from "node:fs"
24
+ import * as path from "node:path"
25
+ import { fileURLToPath } from "node:url"
26
+
27
+ import { activeSessionCountForRepo, maxConcurrentSessionsFromEnv } from "../budget/concurrency.js"
28
+ import { issueShape, issueSizeWarnings, splitIssues, type IssueSizeWarning, type SplitIssue, type WorktreeSpec } from "./split.js"
29
+ import { readCostHistory } from "./cost-history.js"
30
+ import { buildEstimateSection, estimateGroups } from "./cost-estimate.js"
31
+ import {
32
+ loadState,
33
+ loadStateSync,
34
+ mutateState,
35
+ patchGroup,
36
+ saveStateSync,
37
+ updateGroup,
38
+ type OrchestratorGroup,
39
+ type OrchestratorState,
40
+ } from "./state.js"
41
+ import { runReviewWithRetries, type ReviewResult } from "./reviewer.js"
42
+ import { hasRealVerificationActivity, hasRealWorktreeChanges } from "./verification-gate.js"
43
+ import { runFilingStage, runResearchStage } from "./pipeline.js"
44
+ import { analyzeWorktreeSessions } from "./log-analysis.js"
45
+ import { runQaWithRetries, type QaResult } from "../qa/qa.js"
46
+ import { groupWorktreePath, isIterationExhaustion, isPidAlive, isProviderFailure, watchGroups } from "./watch.js"
47
+ import { clearHandoffSummary, readHandoffSummary } from "../engine/handoff.js"
48
+ import { autoSplitOversizedIssues, proposeSemanticSplit } from "./auto-split.js"
49
+ import { DEFAULT_MODEL, OpenRouterClient } from "../llm/openrouter.js"
50
+ import {
51
+ buildStatusSummary,
52
+ formatStatusText,
53
+ isTerminalStatus,
54
+ reconcileGroups,
55
+ verdictLine,
56
+ waitForTerminalState,
57
+ DEFAULT_STATUS_POLL_INTERVAL_MS,
58
+ DEFAULT_STATUS_TIMEOUT_MS,
59
+ type StatusSummary,
60
+ } from "./status.js"
61
+ import { assessGroupCleanupSync, cleanupMain, resolveBaseBranch, type CleanupStatus } from "./cleanup.js"
62
+ import {
63
+ branchSyncStatus,
64
+ ORCHESTRATE_SYNC_DISABLED_ENV,
65
+ syncBranchWithOrigin,
66
+ syncSummaryLines,
67
+ syncWarningLines,
68
+ TRIVIAL_DRIFT_AHEAD,
69
+ } from "./git-sync.js"
70
+ import { resolveModelForMode } from "../config/mode-models.js"
71
+ import { runPreflight, runLocalPreflight, type PreflightResult, type LocalPreflightResult } from "../llm/preflight.js"
72
+ import { resolvePerModeEnv } from "../cli.js"
73
+
74
+ const ORCHESTRATE_USAGE = `headlesscode orchestrate — Phase 2 parallel round
75
+
76
+ Usage:
77
+ headlesscode orchestrate --repo <path> --issue <n> [--issue <n> ...] [options]
78
+ headlesscode orchestrate --repo <path> --issues-json <file> [options]
79
+
80
+ Options:
81
+ --repo <path> Target repo root (required)
82
+ --issue <n> Issue number to include (repeatable)
83
+ --issues-json <file> Read issues from a JSON array of {number,title,body}
84
+ --file-issues With --issues-json: file a REAL GitHub issue for each
85
+ synthetic entry (gh issue create --repo <origin-owner>/<origin-repo>),
86
+ swap in the real number returned, and print one
87
+ confirmation line per created issue. A real, visible
88
+ write to GitHub — opt-in, never automatic.
89
+ Requires --issues-json (nothing to file otherwise)
90
+ --mode <slug> Harness mode for workers (default: code)
91
+ --model <id> Explicit model override for workers + reviewer + QA
92
+ (beats .headlesscode/mode-models.json entries). Without
93
+ it, each role resolves its OWN model from the file:
94
+ worker mode / --review-mode / --qa-mode
95
+ (default: $OPENROUTER_MODEL or the client default)
96
+ --batch <name> Batch id recorded in the state file (default: round-<date>)
97
+ --review-mode <slug> Mode slug used for review sessions (default: deepseek-reviewer)
98
+ --no-review Spawn + watch only; do not run the reviewer
99
+ --qa Run a headless QA session (--mode qa-agent) on each group
100
+ after its review passes; record qa {status,verdict,evidence}
101
+ --qa-mode <slug> Mode slug for QA sessions (default: qa-agent; the target
102
+ repo's .roomodes + .roo/rules-<slug>/ are spliced automatically)
103
+ --deploy After all groups done + reviewed + QA passed, run the
104
+ human-approval deploy gate (scripts/deploy-gate.sh) which
105
+ refuses to run the repo's deploy-production.sh without
106
+ explicit human approval (interactive on a TTY, token/file
107
+ otherwise). Never auto-approves.
108
+ --deploy-args <str> Deploy args forwarded to the deploy script after the gate
109
+ approves (space-separated flags; also DEPLOY_ARGS env)
110
+ --poll-interval-ms <n> Watcher poll interval (default: 5000)
111
+ --memory-dir <path> Phase 3 memory dir for workers (passed as HEADLESSCODE_MEMORY_DIR,
112
+ which run-worker.sh forwards as --memory-dir to each worker CLI)
113
+ --max-concurrent-sessions <n> Phase 6 GLOBAL cap on concurrent sessions across
114
+ processes (default: $HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3).
115
+ When the cap is already reached this run ABORTS with a clear
116
+ message and exit 1 (never queues silently)
117
+ --max-rework-cycles <n> Max automatic rework attempts per group when the review
118
+ finds issues (default: 3). Each rework re-runs a worker on the
119
+ SAME worktree to fix the findings. After the cap, the group is
120
+ marked "needs-human" and left for manual resolution
121
+ --max-iterations <n> Per-session loop iteration cap for EVERY worker in this
122
+ round (default: $HEADLESSCODE_MAX_ITERATIONS or the harness's
123
+ own default, 50). A worker that hits the cap is auto-continued
124
+ on the SAME worktree (see --max-continuations) instead of
125
+ hard-failing the group
126
+ --max-continuations <n> Max automatic continuations per group after a worker
127
+ hits the iteration cap (default: 3). Each continuation
128
+ re-spawns a worker on the SAME worktree with a fresh session.
129
+ After the cap, the group is marked "needs-human" and left for
130
+ manual resolution
131
+ --plan-first Issue #49 experiment: run a SHORT architect-mode
132
+ planning session per worktree BEFORE the code worker, and
133
+ append its plan (PLAN.md) into the worker's task file so the
134
+ code worker executes against it instead of re-discovering
135
+ context (grep/read cycles) from iteration 1. OPT-IN — never
136
+ the default without evidence it helps. The plan session runs
137
+ synchronously inside the spawner, so N plan-first worktrees
138
+ extend the spawn call by ~N × plan-session time
139
+ --plan-first-mode <slug> Mode slug for the plan-first session (default:
140
+ architect; any mode the worker CLI accepts — built-in or
141
+ .roomodes)
142
+ --plan-first-max-iterations <n> Iteration cap for the plan-first session
143
+ (default: 15, or $HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS) —
144
+ deliberately short: the plan session must produce a plan,
145
+ not implement it
146
+ --no-preflight Skip the pre-spawn preflight probe (issue #13): a cheap
147
+ 1-token completion using the EXACT model + provider pin a real
148
+ worker uses, run BEFORE anything is spawned, that distinguishes
149
+ key-invalid / account-balance-exhausted / pinned-provider-down
150
+ / all-clear. ON by default; skip for CI/non-interactive
151
+ contexts that don't want the extra round-trip
152
+ --no-issue-size-check Skip the pre-flight issue-size warning (issue #53): a
153
+ FREE deterministic scan of each issue body that warns loudly
154
+ when it reads like 3+ independent pieces of work (top-level
155
+ numbered/bulleted sections), before anything is spawned — the
156
+ shape that burned iteration caps + budget on real rounds
157
+ . ON by default; the
158
+ warning never aborts the round, so this only silences it
159
+ --no-auto-split Skip auto-split: when an issue is flagged by the size
160
+ check above, an LLM proposes the smallest number of
161
+ independently-shippable sub-issues (grouped by
162
+ deliverable, NOT one per bullet — sequential steps of
163
+ one fix stay together), files them as real GitHub
164
+ issues, and closes the oversized parent with a link
165
+ to each. Fails open (LLM/filing error, or the model
166
+ deciding the issue is genuinely one coherent piece of
167
+ work) by dispatching the original issue unchanged.
168
+ ON by default when a GitHub origin remote is present
169
+ and issueSizeCheck is on; this flag falls back to
170
+ warn-only
171
+ --dry-run Print the split plan, a cost estimate (from recorded
172
+ cost-history, keyed by issue shape — issue #16), and
173
+ the spawn commands; spawn nothing
174
+ --help Show this help and exit
175
+
176
+ Environment:
177
+ HEADLESSCODE_OPENROUTER_API_KEY Required for a real run (workers + reviewer + QA)
178
+ ORCHESTRATOR_MODE Overrides --mode when set
179
+ HEADLESSCODE_CLI Overrides the spawner's CLI command
180
+ HEADLESSCODE_MEMORY_DIR Memory dir inherited by workers (unless --memory-dir overrides)
181
+ DEPLOY_APPROVAL_TOKEN Deploy gate token (non-interactive approval; must match the
182
+ token in <repo>/.deploy-approval)
183
+ DEPLOY_APPROVAL_FILE Path to a one-time approval file created by a human
184
+ (default: <repo>/.worktrees/.deploy-approved-<batch>)
185
+ DEPLOY_ARGS Deploy args forwarded to the deploy script (see --deploy-args)
186
+ HEADLESSCODE_MAX_CONCURRENT_SESSIONS Global concurrency cap (default 3)
187
+ HEADLESSCODE_MAX_COST_USD / HEADLESSCODE_MAX_DURATION_MS
188
+ Per-session budget env fallbacks forwarded to workers
189
+ (run-worker.sh / run-qa.sh pass them as CLI flags)
190
+ HEADLESSCODE_MAX_ITERATIONS
191
+ Per-session iteration cap env fallback forwarded to
192
+ workers as --max-iterations (run-worker.sh)
193
+ `
194
+
195
+ interface OrchestrateOptions {
196
+ repo: string
197
+ issues: number[]
198
+ issuesJson?: string
199
+ /**
200
+ * Fix 3: file a REAL GitHub issue for every synthetic --issues-json entry
201
+ * and substitute the real number before anything else touches `issues`.
202
+ * Opt-in — a real write to a real GitHub repo. Requires --issues-json.
203
+ */
204
+ fileIssues: boolean
205
+ mode: string
206
+ model?: string
207
+ batch?: string
208
+ reviewMode: string
209
+ review: boolean
210
+ /** Phase 4: run a headless QA session per group after review passes. */
211
+ qa: boolean
212
+ /** Phase 4: mode slug for QA sessions (default qa-agent). */
213
+ qaMode: string
214
+ /** Phase 4: run the human-approval deploy gate after everything passes. */
215
+ deploy: boolean
216
+ /** Phase 4: deploy args forwarded to the deploy script after approval. */
217
+ deployArgs?: string
218
+ pollIntervalMs: number
219
+ memoryDir?: string
220
+ /** Phase 6: global concurrent-session cap (default env or 3). */
221
+ maxConcurrentSessions: number
222
+ /** Rework loop: max review-rework attempts per group before giving up (default 3). */
223
+ maxReworkCycles: number
224
+ /**
225
+ * Per-session iteration cap forwarded to every worker (default:
226
+ * $HEADLESSCODE_MAX_ITERATIONS; undefined = leave the harness's own
227
+ * default, 50, untouched).
228
+ */
229
+ maxIterations?: number
230
+ /** Auto-continue: max re-spawns per group after a worker hits the iteration cap (default 3). */
231
+ maxContinuations: number
232
+ /**
233
+ * Issue #49 experiment: run a short architect-mode planning session per
234
+ * worktree BEFORE the code worker, and append its plan (PLAN.md) into the
235
+ * worker's task file so the code worker executes against it instead of
236
+ * re-discovering context from iteration 1. OPT-IN — never the default.
237
+ */
238
+ planFirst: boolean
239
+ /** Mode slug for the plan-first session (default: architect). */
240
+ planFirstMode: string
241
+ /** Iteration cap for the plan-first session (default: 15 — deliberately short). */
242
+ planFirstMaxIterations: number
243
+ /**
244
+ * Issue #13 pre-spawn preflight probe (1-token, exact model + provider
245
+ * pin). ON by default; --no-preflight skips it for CI/non-interactive.
246
+ */
247
+ preflight: boolean
248
+ /**
249
+ * Issue #53 pre-flight issue-size check: a FREE deterministic scan that
250
+ * warns loudly (never aborts) when an issue body reads like 3+ independent
251
+ * pieces of work, before anything is spawned. ON by default;
252
+ * --no-issue-size-check silences it.
253
+ */
254
+ issueSizeCheck: boolean
255
+ /**
256
+ * Auto-split (issue #53 follow-up): when the size check flags an issue,
257
+ * propose a SEMANTIC split (LLM-decided independent deliverables, not a
258
+ * mechanical one-sub-issue-per-bullet explosion — see auto-split.ts) and
259
+ * file the result as real GitHub issues, closing the oversized parent.
260
+ * ON by default when a GitHub 'origin' remote is available; --no-auto-split
261
+ * falls back to warn-only. Requires issueSizeCheck (nothing to act on
262
+ * otherwise) and a real (non---issues-json) round (there is no GitHub
263
+ * issue to close for a synthetic entry).
264
+ */
265
+ autoSplit: boolean
266
+ dryRun: boolean
267
+ help: boolean
268
+ }
269
+
270
+ /**
271
+ * Resolve the default per-session iteration cap from $HEADLESSCODE_MAX_ITERATIONS.
272
+ * undefined (unset or invalid) means "leave the harness's own default (50)
273
+ * untouched" — the harness default itself is never changed, this only makes the
274
+ * round-level cap overridable.
275
+ */
276
+ function maxIterationsFromEnv(): number | undefined {
277
+ const raw = process.env.HEADLESSCODE_MAX_ITERATIONS
278
+ if (raw === undefined || raw === "") {
279
+ return undefined
280
+ }
281
+ const n = Number(raw)
282
+ return Number.isInteger(n) && n > 0 ? n : undefined
283
+ }
284
+
285
+ /**
286
+ * Resolve the plan-first session's iteration cap from
287
+ * $HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS (default: 15 — the plan session is
288
+ * deliberately SHORT: it must produce a plan, not implement it). Invalid
289
+ * values fall back to the default.
290
+ */
291
+ function planFirstMaxIterationsFromEnv(): number {
292
+ const raw = process.env.HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS
293
+ if (raw === undefined || raw === "") {
294
+ return 15
295
+ }
296
+ const n = Number(raw)
297
+ return Number.isInteger(n) && n > 0 ? n : 15
298
+ }
299
+
300
+ export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOptions; error?: string } {
301
+ const options: OrchestrateOptions = {
302
+ repo: "",
303
+ issues: [],
304
+ mode: process.env.ORCHESTRATOR_MODE ?? "code",
305
+ reviewMode: "deepseek-reviewer",
306
+ review: true,
307
+ qa: false,
308
+ qaMode: "qa-agent",
309
+ deploy: false,
310
+ fileIssues: false,
311
+ pollIntervalMs: 5000,
312
+ maxConcurrentSessions: maxConcurrentSessionsFromEnv(),
313
+ maxReworkCycles: 3,
314
+ maxContinuations: 3,
315
+ maxIterations: maxIterationsFromEnv(),
316
+ planFirst: false,
317
+ planFirstMode: process.env.HEADLESSCODE_PLAN_FIRST_MODE ?? "architect",
318
+ planFirstMaxIterations: planFirstMaxIterationsFromEnv(),
319
+ preflight: true,
320
+ issueSizeCheck: true,
321
+ autoSplit: true,
322
+ dryRun: false,
323
+ help: false,
324
+ }
325
+
326
+ for (let i = 0; i < argv.length; i++) {
327
+ const arg = argv[i]
328
+ const eq = arg.indexOf("=")
329
+ const flag = eq === -1 ? arg : arg.slice(0, eq)
330
+ const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
331
+ const next = (): string | undefined => {
332
+ if (inlineValue !== undefined) {
333
+ return inlineValue
334
+ }
335
+ const v = argv[i + 1]
336
+ if (v === undefined || v.startsWith("--")) {
337
+ return undefined
338
+ }
339
+ i++
340
+ return v
341
+ }
342
+
343
+ switch (flag) {
344
+ case "--repo": {
345
+ const v = next()
346
+ if (v === undefined) {
347
+ return { options, error: "Missing value for --repo" }
348
+ }
349
+ options.repo = v
350
+ break
351
+ }
352
+ case "--issue": {
353
+ const v = next()
354
+ const n = v === undefined ? Number.NaN : Number(v)
355
+ if (!Number.isInteger(n) || n <= 0) {
356
+ return { options, error: "--issue requires a positive integer" }
357
+ }
358
+ options.issues.push(n)
359
+ break
360
+ }
361
+ case "--issues-json": {
362
+ const v = next()
363
+ if (v === undefined) {
364
+ return { options, error: "Missing value for --issues-json" }
365
+ }
366
+ options.issuesJson = v
367
+ break
368
+ }
369
+ case "--file-issues":
370
+ options.fileIssues = true
371
+ break
372
+ case "--mode": {
373
+ const v = next()
374
+ if (v === undefined) {
375
+ return { options, error: "Missing value for --mode" }
376
+ }
377
+ options.mode = v
378
+ break
379
+ }
380
+ case "--model": {
381
+ const v = next()
382
+ if (v === undefined) {
383
+ return { options, error: "Missing value for --model" }
384
+ }
385
+ options.model = v
386
+ break
387
+ }
388
+ case "--batch": {
389
+ const v = next()
390
+ if (v === undefined) {
391
+ return { options, error: "Missing value for --batch" }
392
+ }
393
+ options.batch = v
394
+ break
395
+ }
396
+ case "--review-mode": {
397
+ const v = next()
398
+ if (v === undefined) {
399
+ return { options, error: "Missing value for --review-mode" }
400
+ }
401
+ options.reviewMode = v
402
+ break
403
+ }
404
+ case "--poll-interval-ms": {
405
+ const v = next()
406
+ const n = v === undefined ? Number.NaN : Number(v)
407
+ if (!Number.isInteger(n) || n <= 0) {
408
+ return { options, error: "--poll-interval-ms requires a positive integer" }
409
+ }
410
+ options.pollIntervalMs = n
411
+ break
412
+ }
413
+ case "--memory-dir": {
414
+ const v = next()
415
+ if (v === undefined) {
416
+ return { options, error: "Missing value for --memory-dir" }
417
+ }
418
+ options.memoryDir = v
419
+ break
420
+ }
421
+ case "--max-concurrent-sessions": {
422
+ const v = next()
423
+ const n = v === undefined ? Number.NaN : Number(v)
424
+ if (!Number.isInteger(n) || n <= 0) {
425
+ return { options, error: "--max-concurrent-sessions requires a positive integer" }
426
+ }
427
+ options.maxConcurrentSessions = n
428
+ break
429
+ }
430
+ case "--max-rework-cycles": {
431
+ const v = next()
432
+ const n = v === undefined ? Number.NaN : Number(v)
433
+ if (!Number.isInteger(n) || n <= 0) {
434
+ return { options, error: "--max-rework-cycles requires a positive integer" }
435
+ }
436
+ options.maxReworkCycles = n
437
+ break
438
+ }
439
+ case "--max-iterations": {
440
+ const v = next()
441
+ const n = v === undefined ? Number.NaN : Number(v)
442
+ if (!Number.isInteger(n) || n <= 0) {
443
+ return { options, error: "--max-iterations requires a positive integer" }
444
+ }
445
+ options.maxIterations = n
446
+ break
447
+ }
448
+ case "--max-continuations": {
449
+ const v = next()
450
+ const n = v === undefined ? Number.NaN : Number(v)
451
+ if (!Number.isInteger(n) || n <= 0) {
452
+ return { options, error: "--max-continuations requires a positive integer" }
453
+ }
454
+ options.maxContinuations = n
455
+ break
456
+ }
457
+ case "--plan-first":
458
+ options.planFirst = true
459
+ break
460
+ case "--plan-first-mode": {
461
+ const v = next()
462
+ if (v === undefined) {
463
+ return { options, error: "Missing value for --plan-first-mode" }
464
+ }
465
+ options.planFirstMode = v
466
+ break
467
+ }
468
+ case "--plan-first-max-iterations": {
469
+ const v = next()
470
+ const n = v === undefined ? Number.NaN : Number(v)
471
+ if (!Number.isInteger(n) || n <= 0) {
472
+ return { options, error: "--plan-first-max-iterations requires a positive integer" }
473
+ }
474
+ options.planFirstMaxIterations = n
475
+ break
476
+ }
477
+ case "--no-review":
478
+ options.review = false
479
+ break
480
+ case "--qa":
481
+ options.qa = true
482
+ break
483
+ case "--qa-mode": {
484
+ const v = next()
485
+ if (v === undefined) {
486
+ return { options, error: "Missing value for --qa-mode" }
487
+ }
488
+ options.qaMode = v
489
+ break
490
+ }
491
+ case "--deploy":
492
+ options.deploy = true
493
+ break
494
+ case "--deploy-args": {
495
+ // Deploy args legitimately start with "--" (e.g. --local), so
496
+ // bypass next()'s "--"-rejecting value reader.
497
+ const v = inlineValue ?? argv[i + 1]
498
+ if (v === undefined) {
499
+ return { options, error: "Missing value for --deploy-args" }
500
+ }
501
+ i++
502
+ options.deployArgs = v
503
+ break
504
+ }
505
+ case "--no-preflight":
506
+ options.preflight = false
507
+ break
508
+ case "--no-issue-size-check":
509
+ options.issueSizeCheck = false
510
+ break
511
+ case "--no-auto-split":
512
+ options.autoSplit = false
513
+ break
514
+ case "--dry-run":
515
+ options.dryRun = true
516
+ break
517
+ case "--help":
518
+ case "-h":
519
+ options.help = true
520
+ break
521
+ default:
522
+ return { options, error: `Unknown orchestrate argument: ${arg}` }
523
+ }
524
+ }
525
+
526
+ // Fix 3: --file-issues without --issues-json is a usage error — a real
527
+ // --issue <n>/gh-sourced entry already carries a real number, so there is
528
+ // nothing to file. (Match the existing usage-error exit-2 convention.)
529
+ if (options.fileIssues && !options.issuesJson) {
530
+ return { options, error: "--file-issues requires --issues-json (a real --issue <n> round has nothing to file)" }
531
+ }
532
+
533
+ return { options }
534
+ }
535
+
536
+ // ─── Issue loading ───────────────────────────────────────────────────────────
537
+
538
+ function ghAvailable(): boolean {
539
+ try {
540
+ execFileSync("gh", ["--version"], { stdio: "ignore", timeout: 5000 })
541
+ return true
542
+ } catch {
543
+ return false
544
+ }
545
+ }
546
+
547
+ function fetchIssueWithGh(repo: string, number: number): SplitIssue {
548
+ const out = execFileSync("gh", ["issue", "view", String(number), "--json", "number,title,body"], {
549
+ cwd: repo,
550
+ encoding: "utf-8",
551
+ timeout: 30_000,
552
+ })
553
+ const parsed = JSON.parse(out) as { number?: number; title?: string; body?: string | null }
554
+ return { number: parsed.number ?? number, title: parsed.title ?? `issue ${number}`, body: parsed.body ?? undefined }
555
+ }
556
+
557
+ /**
558
+ * Resolve `<owner>/<repo>` from the target repo's `origin` remote (used by
559
+ * --file-issues' `gh issue create --repo <owner>/<repo>`). Accepts both ssh
560
+ * (`git@host:owner/repo.git`) and https (`https://host/owner/repo.git`) URL
561
+ * forms. Returns undefined when there is no origin or it is unparseable.
562
+ */
563
+ function originOwnerRepo(repo: string): string | undefined {
564
+ try {
565
+ const url = execFileSync("git", ["-C", repo, "remote", "get-url", "origin"], {
566
+ encoding: "utf-8",
567
+ timeout: 10_000,
568
+ }).trim()
569
+ const match = url.match(/^(?:git@[^:]+:|https?:\/\/[^/]+\/)([^/]+)\/([^/]+?)(?:\.git)?$/)
570
+ return match ? `${match[1]}/${match[2]}` : undefined
571
+ } catch {
572
+ return undefined
573
+ }
574
+ }
575
+
576
+ /**
577
+ * File one real GitHub issue via `gh issue create` — mirrors fetchIssueWithGh's
578
+ * execFileSync idiom (same cwd/timeout). Returns the real issue number (the
579
+ * trailing number of the issue URL gh prints — parsed rather than depending on
580
+ * a --json flag that only exists in newer gh versions) and its URL.
581
+ */
582
+ function createGhIssue(repo: string, ownerRepo: string, issue: SplitIssue): { number: number; url: string } {
583
+ const out = execFileSync(
584
+ "gh",
585
+ ["issue", "create", "--repo", ownerRepo, "--title", issue.title, "--body", issue.body ?? ""],
586
+ { cwd: repo, encoding: "utf-8", timeout: 30_000 },
587
+ )
588
+ const url = out.trim().split(/\s+/).pop() ?? out.trim()
589
+ const numberMatch = url.match(/\/issues\/(\d+)\/?$/)
590
+ if (!numberMatch) {
591
+ throw new Error(`gh issue create returned an unrecognized response: ${out.trim()}`)
592
+ }
593
+ return { number: Number(numberMatch[1]), url }
594
+ }
595
+
596
+ /**
597
+ * Close a real GitHub issue with a comment (auto-split's parent-issue
598
+ * closeout) — mirrors createGhIssue's execFileSync idiom. `gh issue close
599
+ * --comment` posts the comment atomically with the close, so there's no
600
+ * window where the issue is closed without the sub-issue links, or vice
601
+ * versa.
602
+ */
603
+ function closeGhIssue(repo: string, ownerRepo: string, issueNumber: number, comment: string): void {
604
+ execFileSync("gh", ["issue", "close", String(issueNumber), "--repo", ownerRepo, "--comment", comment], {
605
+ cwd: repo,
606
+ encoding: "utf-8",
607
+ timeout: 30_000,
608
+ })
609
+ }
610
+
611
+ /**
612
+ * Fix 3: file a real GitHub issue for each synthetic issue and substitute the
613
+ * real number, so every downstream step (worktree/branch naming, task files,
614
+ * state persistence, PR-closing comments) uses a number that actually exists
615
+ * on GitHub — no other code needs to know issues were just filed. `shouldFile`
616
+ * gates which entries get filed: orchestrate uses the default (file all —
617
+ * --file-issues requires --issues-json, and every --issues-json entry is
618
+ * synthetic by definition); the predicate exists so tests can cover "entries
619
+ * that already carry a real number pass through untouched" without a real gh
620
+ * call. `createIssue` is injected for the same reason (real: createGhIssue).
621
+ */
622
+ export function fileSyntheticIssues(
623
+ issues: SplitIssue[],
624
+ createIssue: (issue: SplitIssue) => { number: number; url: string },
625
+ shouldFile: (issue: SplitIssue) => boolean = () => true,
626
+ ): { issues: SplitIssue[]; created: Array<{ number: number; title: string; url: string }> } {
627
+ const created: Array<{ number: number; title: string; url: string }> = []
628
+ const next = issues.map((issue) => {
629
+ if (!shouldFile(issue)) {
630
+ return issue
631
+ }
632
+ const real = createIssue(issue)
633
+ created.push({ number: real.number, title: issue.title, url: real.url })
634
+ return { ...issue, number: real.number }
635
+ })
636
+ return { issues: next, created }
637
+ }
638
+
639
+ function loadIssues(options: OrchestrateOptions): SplitIssue[] {
640
+ if (options.issuesJson) {
641
+ const raw = fs.readFileSync(path.resolve(options.issuesJson), "utf-8")
642
+ const parsed = JSON.parse(raw) as unknown
643
+ if (!Array.isArray(parsed)) {
644
+ throw new Error(`--issues-json must be a JSON array of {number,title,body}`)
645
+ }
646
+ return parsed.map((i) => {
647
+ const issue = i as { number?: unknown; title?: unknown; body?: unknown }
648
+ return {
649
+ number: Number(issue.number),
650
+ title: String(issue.title ?? ""),
651
+ body: typeof issue.body === "string" ? issue.body : undefined,
652
+ }
653
+ })
654
+ }
655
+ if (options.issues.length === 0) {
656
+ throw new Error("provide at least one --issue <n> or an --issues-json file")
657
+ }
658
+ if (!ghAvailable()) {
659
+ throw new Error(
660
+ "gh CLI is not available and no --issues-json was given — install gh or pass --issues-json <file>",
661
+ )
662
+ }
663
+ return [...new Set(options.issues)].sort((a, b) => a - b).map((n) => fetchIssueWithGh(options.repo, n))
664
+ }
665
+
666
+ // ─── Task-file generation ────────────────────────────────────────────────────
667
+
668
+ /**
669
+ * Rules splicing happens automatically: addCustomInstructions() (vendored,
670
+ * src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts) reads
671
+ * the central store's `rules/` + `rules-<mode>/` tiers (and the project's
672
+ * `.roo/` equivalents) into the system prompt before the first turn. Task
673
+ * files must NOT tell the model to go read a rules file itself — on a project
674
+ * with no project-local `.roo/rules-code/` (e.g. a fresh `headlesscode init`
675
+ * setup) that read fails outright and burns a consecutive-mistake budget
676
+ * before real work starts. Shared by all five task-file builders below.
677
+ */
678
+ const RULES_SPLICED_NOTE = [
679
+ "Project and mode-specific rules (from the central shared store and this project's own",
680
+ ".roo/rules*) are already spliced into your system prompt automatically — you do not",
681
+ "need to read any rules file yourself.",
682
+ ]
683
+
684
+ /**
685
+ * One issue's title/body keyed by issue number as a string (JSON object keys
686
+ * are always strings). Persisted on OrchestratorGroup at dispatch time so the
687
+ * rework/QA/continuation task files — generated LATER from state alone, when
688
+ * the original issue objects are no longer in hand — can embed the real
689
+ * title/body instead of telling the worker to `gh issue view <n>` (which
690
+ * fails outright for synthetic --issues-json numbers).
691
+ */
692
+ export type IssueBodies = Record<string, { title: string; body?: string }>
693
+
694
+ /**
695
+ * Render each assigned issue as an inline markdown block
696
+ * ("### Issue #<n>: <title>\n\n<body>") for a task file — the issue title/body
697
+ * is already in hand, so nothing needs fetching at session start. An issue
698
+ * whose body was never recorded gets an explicit note; an issue with NO entry
699
+ * at all falls back to the OLD `gh issue view <n>` instruction for that issue
700
+ * only (state written before bodies were persisted — never silently produce an
701
+ * empty assignment section).
702
+ */
703
+ function issueBodyBlocks(
704
+ numbers: number[],
705
+ lookup: (n: number) => { title: string; body?: string } | undefined,
706
+ ): string[] {
707
+ const blocks: string[] = []
708
+ for (const n of numbers) {
709
+ const entry = lookup(n)
710
+ if (entry === undefined) {
711
+ blocks.push(
712
+ `### Issue #${n}`,
713
+ "",
714
+ `No issue body recorded — fetch it yourself: \`gh issue view ${n}\`.`,
715
+ )
716
+ } else {
717
+ blocks.push(
718
+ `### Issue #${n}: ${entry.title}`,
719
+ "",
720
+ entry.body && entry.body.trim() !== ""
721
+ ? entry.body
722
+ : "(no issue body recorded — inspect the codebase and git history for context)",
723
+ )
724
+ }
725
+ }
726
+ return blocks
727
+ }
728
+
729
+ export function buildTaskFileContent(spec: WorktreeSpec, issues: SplitIssue[]): string {
730
+ const assigned = spec.issues.map((n) => `#${n}`).join(", ")
731
+ const issueBlocks = issueBodyBlocks(spec.issues, (n) => issues.find((i) => i.number === n))
732
+ return [
733
+ ...RULES_SPLICED_NOTE,
734
+ "",
735
+ "You are working autonomously in an isolated git worktree, on your own branch, with your own",
736
+ "isolated environment. No human is in the loop until you open a PR. Read this entire file before",
737
+ "doing anything.",
738
+ "",
739
+ "## Your assignment",
740
+ "",
741
+ `GitHub issues assigned to this worktree: ${assigned}. The real title/body of each is inline below.`,
742
+ "",
743
+ ...(issueBlocks.length > 0 ? issueBlocks : ["(no issue numbers recorded — inspect the worktree for context)"]),
744
+ "",
745
+ "## Environment",
746
+ "",
747
+ `Workspace root: this worktree (branch: current branch). All file/command tools are scoped to`,
748
+ "this worktree. Commit per logical unit with the issue number in the message.",
749
+ "",
750
+ "## Scratch",
751
+ "",
752
+ "Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
753
+ "`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
754
+ "worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
755
+ "",
756
+ "## Workflow",
757
+ "",
758
+ "1. Read each assigned issue's inline body above and the relevant source files.",
759
+ "2. Implement the change following the project's own conventions and the rules files.",
760
+ "3. Verify for real (run the project's tests / boot check) and capture the output.",
761
+ "4. Commit with the issue number referenced, one logical change per commit.",
762
+ "5. Push the branch and open a PR (ready, not draft).",
763
+ "6. Post the closing comment below on each issue, then close it.",
764
+ "",
765
+ "## Required comment template",
766
+ "",
767
+ "Your issue-closing comment must use this exact structure (every section present; a section can",
768
+ "be 'N/A' with a reason, but a missing section is itself a review finding):",
769
+ "```",
770
+ "## Issue <n> — Closing Report",
771
+ "",
772
+ "### Files changed",
773
+ "<one line per file>",
774
+ "",
775
+ "### Baseline (before)",
776
+ "<real output>",
777
+ "",
778
+ "### Baseline (after)",
779
+ "<real output>",
780
+ "",
781
+ "### Boot check",
782
+ "<real output or exact reason it didn't apply>",
783
+ "",
784
+ "### Residual risk / notes",
785
+ "<anything not 100% certain>",
786
+ "",
787
+ "PR: <link>",
788
+ "```",
789
+ "",
790
+ "## Closing out",
791
+ "",
792
+ "Once every assigned issue is closed with the template above:",
793
+ "1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
794
+ " numbers, the boot check, and the PR link(s).",
795
+ "2. The harness records your completion automatically (exit code + .harness.done marker) —",
796
+ " no separate phone-home step is needed.",
797
+ "3. Do not start on any issue not assigned to this worktree.",
798
+ "",
799
+ `Scope: ${assigned} only.`,
800
+ ].join("\n")
801
+ }
802
+
803
+ /**
804
+ * Issue #49 plan-first task content: the task file for the SHORT architect-mode
805
+ * planning session that runs INSIDE the fresh worktree before the code worker.
806
+ * The planner must produce a compact implementation plan (written to PLAN.md at
807
+ * the workspace root, overwriting any previous plan) and stop — never
808
+ * implement, never edit source, never ask the human, never switch modes. The
809
+ * spawner appends PLAN.md into ORCHESTRATOR_TASK.md so the code worker that
810
+ * follows executes against it. Reuses the other task files' conventions (rules
811
+ * preamble, scope line).
812
+ */
813
+ export function buildPlanFirstTaskFileContent(
814
+ group: { name: string; issues?: number[] },
815
+ /** The in-hand issues (their title/body is embedded inline). Optional for
816
+ * backward compatibility with callers that only have persisted state. */
817
+ issues?: SplitIssue[],
818
+ ): string {
819
+ const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
820
+ const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => issues?.find((i) => i.number === n))
821
+ return [
822
+ ...RULES_SPLICED_NOTE,
823
+ "",
824
+ "You are the PLANNING phase of an autonomous two-phase workflow, running in an isolated git",
825
+ "worktree on your own branch. Your ONLY job is to produce a compact, actionable implementation",
826
+ "plan for the code session that follows. Read this entire file before doing anything.",
827
+ "",
828
+ "## Your assignment (planning only — DO NOT implement)",
829
+ "",
830
+ `Assigned issues: ${assigned}. The real title/body of each is inline below.`,
831
+ "",
832
+ ...(issueBlocks.length > 0
833
+ ? issueBlocks
834
+ : ["(no issue numbers recorded — read ORCHESTRATOR_TASK.md for the full task)"]),
835
+ "",
836
+ "1. Read `ORCHESTRATOR_TASK.md` at the workspace root — it is the FULL task the code session",
837
+ " must execute. Read the assigned issues' inline bodies above.",
838
+ "2. Explore the codebase just enough to ground the plan (grep/read the files the issues touch).",
839
+ "3. Write the plan to `PLAN.md` at the workspace root (OVERWRITE any existing PLAN.md) as a",
840
+ " compact markdown document: goal, concrete steps (files + functions), verification steps",
841
+ " (tests/boot), and any risks. A code session will execute it — make every step something",
842
+ " another session can act on without re-doing your exploration.",
843
+ "4. Finish with attempt_completion summarizing where the plan was written.",
844
+ "",
845
+ "## Scratch",
846
+ "",
847
+ "Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
848
+ "`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
849
+ "worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
850
+ "",
851
+ "## Constraints",
852
+ "",
853
+ "- DO NOT implement anything: no source edits, no commits, no pushes, no PRs.",
854
+ "- DO NOT ask the human questions — you are headless; decide autonomously and write the plan.",
855
+ "- DO NOT use switch_mode — this session ends with attempt_completion, period.",
856
+ "- Keep the plan SHORT (a compact plan beats a tome; the code session re-verifies anything it",
857
+ " doubts before trusting it).",
858
+ "",
859
+ `Scope: ${assigned} only — planning phase.`,
860
+ ].join("\n")
861
+ }
862
+
863
+ export function writeTaskFiles(
864
+ repo: string,
865
+ specs: WorktreeSpec[],
866
+ issues: SplitIssue[],
867
+ opts: { planFirst?: boolean } = {},
868
+ ): string[] {
869
+ const dir = path.join(repo, "plans", "parallel-tasks")
870
+ fs.mkdirSync(dir, { recursive: true })
871
+ const written: string[] = []
872
+ for (const spec of specs) {
873
+ const target = path.join(dir, spec.taskFile)
874
+ fs.writeFileSync(target, buildTaskFileContent(spec, issues), "utf-8")
875
+ written.push(target)
876
+ if (opts.planFirst) {
877
+ // Issue #49: the plan-first session's task file (the spawner copies
878
+ // it into the worktree as ORCHESTRATOR_PLAN.md and runs it BEFORE
879
+ // the code worker). Named `<name>-plan.md` — the spawner derives it
880
+ // from the worktree name alone, so a standalone spawner invocation
881
+ // can fall back to a generic planner prompt when it is absent.
882
+ const planTarget = path.join(dir, `${spec.name}-plan.md`)
883
+ fs.writeFileSync(planTarget, buildPlanFirstTaskFileContent(spec, issues), "utf-8")
884
+ written.push(planTarget)
885
+ }
886
+ }
887
+ return written
888
+ }
889
+
890
+ /**
891
+ * Rework-loop task content (see plans/rework-loop.md): a NEW task file for a
892
+ * group whose review came back with a "finding" verdict. Reuses the original
893
+ * task file's structure/conventions (rules preamble, closing-report template,
894
+ * scope) but the assignment is the reviewer's `pending_review_findings` list
895
+ * rather than the raw issues. The worker is expected to address EVERY finding,
896
+ * re-verify for real, and re-close/re-comment with fresh evidence per the
897
+ * closing-report template — same contract as the first attempt.
898
+ */
899
+ export function buildReworkTaskFileContent(
900
+ group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
901
+ findings: string[],
902
+ reworkCount: number,
903
+ /** Persisted issue title/body captured at dispatch time (see
904
+ * OrchestratorGroup.issueBodies). Absent for state written before the
905
+ * bodies feature — falls back to the old `gh issue view <n>` instruction. */
906
+ issueBodies?: IssueBodies,
907
+ ): string {
908
+ const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
909
+ const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
910
+ const findingsList = findings.length > 0 ? findings.map((f, i) => `${i + 1}. ${f}`).join("\n") : ""
911
+ return [
912
+ ...RULES_SPLICED_NOTE,
913
+ "",
914
+ "You are working autonomously in the SAME isolated git worktree as the previous attempt, on",
915
+ "the same branch. The automated adversarial reviewer found problems with that attempt. Read",
916
+ "this entire file before doing anything.",
917
+ "",
918
+ `## Rework cycle ${reworkCount} — address the review findings`,
919
+ "",
920
+ `Assigned issues: ${assigned}. The real title/body of each is inline below.`,
921
+ "",
922
+ ...(issueBlocks.length > 0
923
+ ? issueBlocks
924
+ : ["(no issue numbers recorded — inspect the worktree for context)"]),
925
+ "",
926
+ "## Review findings to fix (from the automated review)",
927
+ "",
928
+ ...findingsList.split("\n"),
929
+ ...(findingsList === "" ? ["(no specific findings were recorded — re-read the previous attempt's", "diff and re-verify everything the reviewer might have flagged.)"] : []),
930
+ "",
931
+ "Your job: address EVERY finding above. Do not skip any. Do not claim a finding is fixed",
932
+ "without reproducing the fix for real.",
933
+ "",
934
+ "## Scratch",
935
+ "",
936
+ "Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
937
+ "`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
938
+ "worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
939
+ "",
940
+ "## Workflow",
941
+ "",
942
+ "1. Read each finding above and the relevant source files.",
943
+ "2. Fix every finding, following the project's own conventions and the rules files.",
944
+ "3. Re-verify for real (run the project's tests / boot check) and capture the output —",
945
+ " output in a report is a claim, not evidence, until reproduced.",
946
+ "4. Commit with the issue number referenced, one logical change per commit.",
947
+ "5. Push the branch and update the PR (or open a new one if needed).",
948
+ "6. Post an UPDATED closing comment below on each issue using the required template, then",
949
+ " re-close it (or leave it open with the updated report if it must stay open).",
950
+ "",
951
+ "## Required comment template",
952
+ "",
953
+ "Your issue-closing comment must use this exact structure (every section present; a section can",
954
+ "be 'N/A' with a reason, but a missing section is itself a review finding):",
955
+ "```",
956
+ `## Issue <n> — Closing Report (rework cycle ${reworkCount})`,
957
+ "",
958
+ "### Files changed",
959
+ "<one line per file>",
960
+ "",
961
+ "### Baseline (before)",
962
+ "<real output>",
963
+ "",
964
+ "### Baseline (after)",
965
+ "<real output>",
966
+ "",
967
+ "### Boot check",
968
+ "<real output or exact reason it didn't apply>",
969
+ "",
970
+ "### Residual risk / notes",
971
+ "<anything not 100% certain>",
972
+ "",
973
+ "PR: <link>",
974
+ "```",
975
+ "",
976
+ "## Closing out",
977
+ "",
978
+ "Once every finding is addressed and every assigned issue is re-closed with fresh evidence:",
979
+ "1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
980
+ " numbers, the boot check, and the PR link(s).",
981
+ "2. The harness records your completion automatically (exit code + .harness.done marker) —",
982
+ " no separate phone-home step is needed.",
983
+ "3. Do not start on any issue not assigned to this worktree.",
984
+ "",
985
+ `Scope: ${assigned} only — rework cycle ${reworkCount}.`,
986
+ ].join("\n")
987
+ }
988
+
989
+ /**
990
+ * QA-fail rework task content (issue #52): a NEW task file for a group whose
991
+ * QA session came back with a real "fail" verdict (review had passed). Same
992
+ * structure/conventions as buildReworkTaskFileContent, but the assignment is
993
+ * the QA evidence — what the QA session actually found and verified — rather
994
+ * than the reviewer's findings list. The worker is expected to address every
995
+ * piece of evidence, re-verify for real, and re-close with fresh evidence per
996
+ * the closing-report template — same contract as the first attempt.
997
+ */
998
+ export function buildQaReworkTaskFileContent(
999
+ group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
1000
+ evidence: string,
1001
+ reworkCount: number,
1002
+ reportPath?: string,
1003
+ /** Persisted issue title/body captured at dispatch time (see
1004
+ * OrchestratorGroup.issueBodies). Absent for state written before the
1005
+ * bodies feature — falls back to the old `gh issue view <n>` instruction. */
1006
+ issueBodies?: IssueBodies,
1007
+ ): string {
1008
+ const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
1009
+ const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
1010
+ // Issue #34-style: point at the QA session's COMPLETE final report so the
1011
+ // full reasoning is one file-read away (evidence is only a 4000-char slice).
1012
+ const reportLines = reportPath
1013
+ ? [
1014
+ "",
1015
+ `The full QA session report is at: \`${reportPath}\` — read it for the complete`,
1016
+ "reasoning behind every finding before you start.",
1017
+ ]
1018
+ : []
1019
+ return [
1020
+ ...RULES_SPLICED_NOTE,
1021
+ "",
1022
+ "You are working autonomously in the SAME isolated git worktree as the previous attempt, on",
1023
+ "the same branch. Review passed, but the automated QA session found real problems with the",
1024
+ "attempt — treat them as NEW WORK. Read this entire file before doing anything.",
1025
+ "",
1026
+ `## Rework cycle ${reworkCount} — address the QA findings`,
1027
+ "",
1028
+ `Assigned issues: ${assigned}. The real title/body of each is inline below.`,
1029
+ "",
1030
+ ...(issueBlocks.length > 0
1031
+ ? issueBlocks
1032
+ : ["(no issue numbers recorded — inspect the worktree for context)"]),
1033
+ "",
1034
+ "## QA findings to fix (from the automated QA)",
1035
+ "",
1036
+ ...(evidence === ""
1037
+ ? [
1038
+ "(the QA report recorded no extractable evidence — re-run the QA checklist against the",
1039
+ "worktree yourself and fix whatever fails.)",
1040
+ ]
1041
+ : [evidence]),
1042
+ ...reportLines,
1043
+ "",
1044
+ "Your job: address EVERY finding above. Do not skip any. Do not claim a finding is fixed",
1045
+ "without reproducing the fix for real.",
1046
+ "",
1047
+ "## Scratch",
1048
+ "",
1049
+ "Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
1050
+ "`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
1051
+ "worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
1052
+ "",
1053
+ "## Workflow",
1054
+ "",
1055
+ "1. Read each finding above and the relevant source files.",
1056
+ "2. Fix every finding, following the project's own conventions and the rules files.",
1057
+ "3. Re-verify for real (run the project's tests / boot check) and capture the output —",
1058
+ " output in a report is a claim, not evidence, until reproduced.",
1059
+ "4. Commit with the issue number referenced, one logical change per commit.",
1060
+ "5. Push the branch and update the PR (or open a new one if needed).",
1061
+ "6. Post an UPDATED closing comment below on each issue using the required template, then",
1062
+ " re-close it (or leave it open with the updated report if it must stay open).",
1063
+ "",
1064
+ "## Required comment template",
1065
+ "",
1066
+ "Your issue-closing comment must use this exact structure (every section present; a section can",
1067
+ "be 'N/A' with a reason, but a missing section is itself a review finding):",
1068
+ "```",
1069
+ `## Issue <n> — Closing Report (rework cycle ${reworkCount})`,
1070
+ "",
1071
+ "### Files changed",
1072
+ "<one line per file>",
1073
+ "",
1074
+ "### Baseline (before)",
1075
+ "<real output>",
1076
+ "",
1077
+ "### Baseline (after)",
1078
+ "<real output>",
1079
+ "",
1080
+ "### Boot check",
1081
+ "<real output or exact reason it didn't apply>",
1082
+ "",
1083
+ "### Residual risk / notes",
1084
+ "<anything not 100% certain>",
1085
+ "",
1086
+ "PR: <link>",
1087
+ "```",
1088
+ "",
1089
+ "## Closing out",
1090
+ "",
1091
+ "Once every finding is addressed and every assigned issue is re-closed with fresh evidence:",
1092
+ "1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
1093
+ " numbers, the boot check, and the PR link(s).",
1094
+ "2. The harness records your completion automatically (exit code + .harness.done marker) —",
1095
+ " no separate phone-home step is needed.",
1096
+ "3. Do not start on any issue not assigned to this worktree.",
1097
+ "",
1098
+ `Scope: ${assigned} only — rework cycle ${reworkCount}.`,
1099
+ ].join("\n")
1100
+ }
1101
+
1102
+ /**
1103
+ * Continuation task content (plans/issues/02-orchestrate-iteration-plumbing.md,
1104
+ * Part 2): a NEW task file for a group whose worker exited because it hit the
1105
+ * iteration cap. The worktree is untouched and resumable — the partial edits
1106
+ * are on disk — so the next session picks up where the last left off. Reuses
1107
+ * the original task file's conventions (rules preamble, closing-report
1108
+ * template, scope) but the assignment is "keep going" rather than the raw
1109
+ * issues.
1110
+ */
1111
+ export function buildContinuationTaskFileContent(
1112
+ group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
1113
+ continuationCount: number,
1114
+ /** Persisted issue title/body captured at dispatch time (see
1115
+ * OrchestratorGroup.issueBodies). Absent for state written before the
1116
+ * bodies feature — falls back to the old `gh issue view <n>` instruction. */
1117
+ issueBodies?: IssueBodies,
1118
+ /** The previous session's own condensed summary of its history (see
1119
+ * src/engine/handoff.ts) — file reads, command outputs, decisions made —
1120
+ * so this cycle doesn't have to re-derive them from scratch. Absent when
1121
+ * the previous session's handoff write failed/didn't run (older state,
1122
+ * or a very short session); the git-status-inspection instruction below
1123
+ * is the fallback either way, not a redundant step. */
1124
+ handoffSummary?: string,
1125
+ ): string {
1126
+ const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
1127
+ const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
1128
+ return [
1129
+ ...RULES_SPLICED_NOTE,
1130
+ "",
1131
+ "You are working autonomously in the SAME isolated git worktree as the previous session(s), on",
1132
+ "the same branch. The previous session was cut short — either it hit the harness's iteration cap, or",
1133
+ "it hit a transient LLM-provider failure (a hard-pinned model with no fallback provider had a bad",
1134
+ "moment; not a fault in your task or the code). Neither is a review verdict — its partial work is",
1135
+ "still on disk: uncommitted edits, commits, notes — pick up exactly where it left off. Read this",
1136
+ "entire file before doing anything.",
1137
+ "",
1138
+ `## Continuation cycle ${continuationCount} — complete the task`,
1139
+ "",
1140
+ `Assigned issues: ${assigned}. The real title/body of each is inline below.`,
1141
+ "",
1142
+ ...(issueBlocks.length > 0
1143
+ ? issueBlocks
1144
+ : ["(no issue numbers recorded — inspect the worktree for context)"]),
1145
+ "",
1146
+ "## What the previous session already learned — do not re-derive this",
1147
+ "",
1148
+ ...(handoffSummary
1149
+ ? [
1150
+ "The previous session summarized its own history before it ran out of iterations. Trust it and",
1151
+ "build on it — re-reading files or re-running commands it already covered wastes iterations you",
1152
+ "need for finishing the actual task. It CAN be wrong (compression loses detail, and the session's",
1153
+ "own notes may be incomplete) — if something below conflicts with what you observe on disk, what",
1154
+ "you observe wins, but don't re-verify things it reports with confidence just to be safe.",
1155
+ "",
1156
+ "```",
1157
+ handoffSummary.trim(),
1158
+ "```",
1159
+ ]
1160
+ : [
1161
+ "(No handoff summary available for this cycle — the previous session's write either failed or",
1162
+ "never ran. Fall back to inspecting the worktree yourself, below.)",
1163
+ ]),
1164
+ "",
1165
+ "## What the previous session did",
1166
+ "",
1167
+ "- Inspect the worktree's git status, recent commits, and uncommitted changes to see exactly",
1168
+ " how far the previous session got — this is a real-state CHECK, not your primary source of",
1169
+ " context; the summary above should already tell you most of this.",
1170
+ "- Do NOT restart from scratch: continue the existing work toward the issues above.",
1171
+ "- If the previous session left a partial fix, finish it and verify it for real.",
1172
+ "",
1173
+ "## Scratch",
1174
+ "",
1175
+ "Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
1176
+ "`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
1177
+ "worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
1178
+ "",
1179
+ "## Workflow",
1180
+ "",
1181
+ "1. Read each assigned issue's inline body above and the relevant source files.",
1182
+ "2. Continue the implementation following the project's own conventions and the rules files.",
1183
+ "3. Verify for real (run the project's tests / boot check) and capture the output —",
1184
+ " output in a report is a claim, not evidence, until reproduced.",
1185
+ "4. Commit with the issue number referenced, one logical change per commit.",
1186
+ "5. Push the branch and open a PR (ready, not draft).",
1187
+ "6. Post the closing comment below on each issue using the required template, then close it.",
1188
+ "",
1189
+ "## Required comment template",
1190
+ "",
1191
+ "Your issue-closing comment must use this exact structure (every section present; a section can",
1192
+ "be 'N/A' with a reason, but a missing section is itself a review finding):",
1193
+ "```",
1194
+ `## Issue <n> — Closing Report (continuation cycle ${continuationCount})`,
1195
+ "",
1196
+ "### Files changed",
1197
+ "<one line per file>",
1198
+ "",
1199
+ "### Baseline (before)",
1200
+ "<real output>",
1201
+ "",
1202
+ "### Baseline (after)",
1203
+ "<real output>",
1204
+ "",
1205
+ "### Boot check",
1206
+ "<real output or exact reason it didn't apply>",
1207
+ "",
1208
+ "### Residual risk / notes",
1209
+ "<anything not 100% certain>",
1210
+ "",
1211
+ "PR: <link>",
1212
+ "```",
1213
+ "",
1214
+ "## Closing out",
1215
+ "",
1216
+ "Once every assigned issue is closed with the template above:",
1217
+ "1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
1218
+ " numbers, the boot check, and the PR link(s).",
1219
+ "2. The harness records your completion automatically (exit code + .harness.done marker) —",
1220
+ " no separate phone-home step is needed.",
1221
+ "3. Do not start on any issue not assigned to this worktree.",
1222
+ "",
1223
+ `Scope: ${assigned} only — continuation cycle ${continuationCount}.`,
1224
+ ].join("\n")
1225
+ }
1226
+
1227
+ // ─── Rework-loop decision ────────────────────────────────────────────────────
1228
+
1229
+ /**
1230
+ * The outcome of deciding what to do with a group whose review came back with
1231
+ * a "finding" verdict (see plans/rework-loop.md). Either the group gets a NEW
1232
+ * rework attempt on the SAME worktree (shouldSpawn), or it has exhausted its
1233
+ * `--max-rework-cycles` budget and is marked for human attention.
1234
+ */
1235
+ export interface ReworkDecision {
1236
+ /** State patch to apply to the group (rework reset, or needs-human). */
1237
+ patch: Partial<Omit<OrchestratorGroup, "name">>
1238
+ /** Whether a rework worker should be spawned on the same worktree. */
1239
+ shouldSpawn: boolean
1240
+ /** The group's reworkCount after this decision. */
1241
+ newReworkCount: number
1242
+ /** Absolute path of the rework task file (only when shouldSpawn). */
1243
+ taskFilePath?: string
1244
+ /** Content of the rework task file (only when shouldSpawn). */
1245
+ taskContent?: string
1246
+ /** Exact `bash -c` command that launches the rework worker (only when shouldSpawn). */
1247
+ spawnCommand?: string
1248
+ }
1249
+
1250
+ /**
1251
+ * Decide what to do after a review "finding" verdict, and build the pieces the
1252
+ * caller needs (state patch, task file, spawn command). Extracted from the
1253
+ * watchGroups callback so every branch is unit-testable without a live round.
1254
+ *
1255
+ * - Below the cap: the group is reset to "running" on the SAME worktree with a
1256
+ * fresh `spawned` timestamp (so the watcher's stall guard measures the rework
1257
+ * worker, not the original one), review/QA records are cleared, reworkCount is
1258
+ * incremented, and a rework task file + worker launch command are produced.
1259
+ * - At the cap: no spawn; the group is marked `needs-human` (a TERMINAL state,
1260
+ * deliberately distinct from `failed`) with the findings left recorded so a
1261
+ * human can see exactly what the reviewer flagged.
1262
+ */
1263
+ export function handleReviewVerdict(
1264
+ group: OrchestratorGroup,
1265
+ result: ReviewResult,
1266
+ repo: string,
1267
+ maxReworkCycles: number,
1268
+ mode: string,
1269
+ model?: string,
1270
+ /** Per-session iteration cap forwarded to the rework worker (default: harness's own). */
1271
+ maxIterations?: number,
1272
+ ): ReworkDecision {
1273
+ const currentRework = typeof group.reworkCount === "number" ? group.reworkCount : 0
1274
+ const newReworkCount = currentRework + 1
1275
+ const now = new Date().toISOString()
1276
+
1277
+ if (currentRework < maxReworkCycles) {
1278
+ const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-rework${newReworkCount}.md`)
1279
+ const taskContent = buildReworkTaskFileContent(group, result.findings, newReworkCount, group.issueBodies)
1280
+ const worktreePath = groupWorktreePath(repo, group)
1281
+ const spawnCommand =
1282
+ `bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
1283
+ `"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
1284
+ (model ? ` --model "${model}"` : "") +
1285
+ (maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
1286
+ return {
1287
+ shouldSpawn: true,
1288
+ newReworkCount,
1289
+ taskFilePath,
1290
+ taskContent,
1291
+ spawnCommand,
1292
+ patch: {
1293
+ status: "running",
1294
+ reworkCount: newReworkCount,
1295
+ // Fresh spawned timestamp: the stall guard must measure the
1296
+ // rework worker, not the (possibly hours-old) original attempt.
1297
+ spawned: now,
1298
+ // A fresh cycle needs a fresh review — clear the old verdict,
1299
+ // findings, QA, and the previous attempt's completion artifacts.
1300
+ review_verdict: undefined,
1301
+ pending_review_findings: undefined,
1302
+ reviewed_at: undefined,
1303
+ review_report: undefined,
1304
+ qa: undefined,
1305
+ stalled: undefined,
1306
+ exit_code: undefined,
1307
+ summary: undefined,
1308
+ // Must also clear cost_recorded: the rework worker adds MORE real
1309
+ // cost on this worktree, and recordCostIfSettled's one-shot gate
1310
+ // would otherwise silently skip re-recording the combined total
1311
+ // once this cycle re-settles — a real bug caught while manually
1312
+ // recovering a stuck round (see cost-history.ts).
1313
+ cost_recorded: undefined,
1314
+ last_activity: {
1315
+ note: `rework cycle ${newReworkCount}: re-spawned worker on the same worktree to fix review findings`,
1316
+ },
1317
+ },
1318
+ }
1319
+ }
1320
+
1321
+ // Cap reached: no new attempt. Terminal "needs-human" outcome — the
1322
+ // findings stay recorded so a human can see exactly what failed.
1323
+ return {
1324
+ shouldSpawn: false,
1325
+ newReworkCount: currentRework,
1326
+ patch: {
1327
+ status: "needs-human",
1328
+ reworkCount: currentRework,
1329
+ last_activity: {
1330
+ note: `exhausted ${currentRework} rework attempt(s); review still has ${result.findings.length} finding(s) — needs human`,
1331
+ },
1332
+ },
1333
+ }
1334
+ }
1335
+
1336
+ /**
1337
+ * QA-side twin of handleReviewVerdict (issue #52): a REAL QA "fail" verdict
1338
+ * (as opposed to a QA-SESSION "error" — see handleQaSessionError) is
1339
+ * actionable new work, exactly like a review "finding". Below the
1340
+ * `--max-rework-cycles` cap the group is reset to "running" on the SAME
1341
+ * worktree with a fresh `spawned` timestamp, review/QA/cost artifacts are
1342
+ * cleared, reworkCount is incremented, and a QA rework task file (fed from
1343
+ * `qa.evidence` rather than review findings) + worker launch command are
1344
+ * produced. At the cap: no spawn; the group is marked `needs-human` with the
1345
+ * QA verdict/evidence left recorded so a human can see exactly what failed.
1346
+ */
1347
+ export function handleQaVerdict(
1348
+ group: OrchestratorGroup,
1349
+ result: QaResult,
1350
+ repo: string,
1351
+ maxReworkCycles: number,
1352
+ mode: string,
1353
+ model?: string,
1354
+ /** Per-session iteration cap forwarded to the rework worker (default: harness's own). */
1355
+ maxIterations?: number,
1356
+ ): ReworkDecision {
1357
+ const currentRework = typeof group.reworkCount === "number" ? group.reworkCount : 0
1358
+ const newReworkCount = currentRework + 1
1359
+ const now = new Date().toISOString()
1360
+
1361
+ if (currentRework < maxReworkCycles) {
1362
+ const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-qa-rework${newReworkCount}.md`)
1363
+ const taskContent = buildQaReworkTaskFileContent(
1364
+ group,
1365
+ result.evidence,
1366
+ newReworkCount,
1367
+ result.reportPath,
1368
+ group.issueBodies,
1369
+ )
1370
+ const worktreePath = groupWorktreePath(repo, group)
1371
+ const spawnCommand =
1372
+ `bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
1373
+ `"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
1374
+ (model ? ` --model "${model}"` : "") +
1375
+ (maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
1376
+ return {
1377
+ shouldSpawn: true,
1378
+ newReworkCount,
1379
+ taskFilePath,
1380
+ taskContent,
1381
+ spawnCommand,
1382
+ patch: {
1383
+ status: "running",
1384
+ reworkCount: newReworkCount,
1385
+ // Fresh spawned timestamp: the stall guard must measure the
1386
+ // rework worker, not the (possibly hours-old) original attempt.
1387
+ spawned: now,
1388
+ // The rework worker changes code, so review AND QA must both
1389
+ // re-run — same fresh-cycle reset as handleReviewVerdict.
1390
+ review_verdict: undefined,
1391
+ pending_review_findings: undefined,
1392
+ reviewed_at: undefined,
1393
+ review_report: undefined,
1394
+ qa: undefined,
1395
+ stalled: undefined,
1396
+ exit_code: undefined,
1397
+ summary: undefined,
1398
+ // The rework worker adds MORE real cost on this worktree; the
1399
+ // one-shot recordCostIfSettled gate must re-record once this
1400
+ // cycle re-settles (same reasoning as handleReviewVerdict).
1401
+ cost_recorded: undefined,
1402
+ last_activity: {
1403
+ note: `rework cycle ${newReworkCount}: re-spawned worker on the same worktree to fix QA findings`,
1404
+ },
1405
+ },
1406
+ }
1407
+ }
1408
+
1409
+ // Cap reached: no new attempt. Terminal "needs-human" outcome — the QA
1410
+ // verdict/evidence stay recorded so a human can see exactly what failed.
1411
+ return {
1412
+ shouldSpawn: false,
1413
+ newReworkCount: currentRework,
1414
+ patch: {
1415
+ status: "needs-human",
1416
+ reworkCount: currentRework,
1417
+ qa: {
1418
+ status: "failed",
1419
+ verdict: result.verdict,
1420
+ evidence: result.evidence.slice(0, 4000),
1421
+ report: result.reportPath,
1422
+ updated: now,
1423
+ },
1424
+ last_activity: {
1425
+ note: `exhausted ${currentRework} rework attempt(s); QA still failing — needs human`,
1426
+ },
1427
+ },
1428
+ }
1429
+ }
1430
+
1431
+ /**
1432
+ * A review-SESSION failure (crash/budget/mistake-limit — see
1433
+ * runReviewWithRetries) survived every retry: verdict "error", distinct from
1434
+ * a real code "finding". This must NEVER feed into handleReviewVerdict's
1435
+ * rework-a-worker path — there is nothing actionable in a placeholder error
1436
+ * message, and spawning a worker to "fix" it wastes a full session for
1437
+ * nothing (caught live 2026-08-05 on issue #17's own round: a review
1438
+ * session's own bounded-failure triggered a pointless worker rework cycle).
1439
+ * Always terminal needs-human — a human should look at why review sessions
1440
+ * keep failing against this worktree, not have a worker respawned blindly.
1441
+ */
1442
+ export function handleReviewSessionError(result: ReviewResult): Partial<OrchestratorGroup> {
1443
+ return {
1444
+ status: "needs-human",
1445
+ review_verdict: result.verdict,
1446
+ pending_review_findings: result.findings,
1447
+ reviewed_at: new Date().toISOString(),
1448
+ last_activity: { note: `review session kept failing: ${result.summary}` },
1449
+ }
1450
+ }
1451
+
1452
+ /**
1453
+ * QA-side twin of handleReviewSessionError: a QA-SESSION failure (verdict
1454
+ * "error" — see runQaWithRetries) survived every retry. Must escalate to
1455
+ * needs-human directly, never recorded as an ordinary "failed" QA (which
1456
+ * would settle the round with cost recorded and nothing left to retry,
1457
+ * while QA never actually validated anything). Caught live 2026-08-05 on
1458
+ * issue #18's own round.
1459
+ */
1460
+ export function handleQaSessionError(result: QaResult): Partial<OrchestratorGroup> {
1461
+ return {
1462
+ status: "needs-human",
1463
+ qa: {
1464
+ status: "failed",
1465
+ verdict: result.verdict,
1466
+ evidence: result.evidence.slice(0, 4000),
1467
+ updated: new Date().toISOString(),
1468
+ },
1469
+ last_activity: { note: `QA session kept failing: ${result.summary}` },
1470
+ }
1471
+ }
1472
+
1473
+ // ─── Iteration-exhaustion continuation decision ──────────────────────────────
1474
+
1475
+ /**
1476
+ * The outcome of deciding what to do with a group whose worker failed with
1477
+ * exit != 0 (plans/issues/02-orchestrate-iteration-plumbing.md, Part 2).
1478
+ * Either the group gets a NEW continuation session on the SAME worktree
1479
+ * (shouldSpawn), or it has exhausted its `--max-continuations` budget and is
1480
+ * marked for human attention, or the failure is not the iteration-cap reason
1481
+ * at all and the group stays `failed`.
1482
+ */
1483
+ export interface ContinuationDecision {
1484
+ /** State patch to apply to the group (running reset, needs-human, or empty for a real failure). */
1485
+ patch: Partial<Omit<OrchestratorGroup, "name">>
1486
+ /** Whether a continuation worker should be spawned on the same worktree. */
1487
+ shouldSpawn: boolean
1488
+ /** The group's continuationCount after this decision. */
1489
+ newContinuationCount: number
1490
+ /** Absolute path of the continuation task file (only when shouldSpawn). */
1491
+ taskFilePath?: string
1492
+ /** Content of the continuation task file (only when shouldSpawn). */
1493
+ taskContent?: string
1494
+ /** Exact `bash -c` command that launches the continuation worker (only when shouldSpawn). */
1495
+ spawnCommand?: string
1496
+ }
1497
+
1498
+ /**
1499
+ * Decide what to do after a worker failed with exit != 0, and build the pieces
1500
+ * the caller needs (state patch, task file, spawn command). Extracted from the
1501
+ * watchGroups callback so every branch is unit-testable without a live round.
1502
+ *
1503
+ * - The failure is NEITHER the iteration-cap reason NOR a transient
1504
+ * LLM-provider failure (budget stop, tool-error storm, a real code/task
1505
+ * bug — or a clean exit with either string merely in the log tail): no
1506
+ * continuation; the patch is empty and the watcher's "failed" stands.
1507
+ * Provider failures earned the same auto-continue treatment as the
1508
+ * iteration cap after a live incident (2026-08-08): a hard-pinned model
1509
+ * with no fallback provider hit an HTTP 520 mid-session, the session
1510
+ * died outright (this predates src/engine/loop.ts's callMainLlm retry —
1511
+ * even with that retry, a longer outage still exhausts it), and the
1512
+ * group sat in the generic terminal "failed" state — indistinguishable
1513
+ * from a real bug — until a human happened to read harness.log. Bounded
1514
+ * the same way as the iteration cap (maxContinuations), so a genuinely
1515
+ * dead provider still surfaces to a human eventually, it just gets a
1516
+ * few free retries first instead of zero.
1517
+ * - Below the cap: the group is reset to "running" on the SAME worktree with a
1518
+ * fresh `spawned` timestamp (so the stall guard measures the continuation
1519
+ * worker, not the attempt that hit the cap), continuationCount is
1520
+ * incremented, completion artifacts are cleared, and a continuation task
1521
+ * file + worker launch command are produced.
1522
+ * - At the cap: no spawn; the group is marked `needs-human` (a TERMINAL state,
1523
+ * deliberately distinct from `failed`).
1524
+ */
1525
+ export function handleIterationExhaustion(
1526
+ group: OrchestratorGroup,
1527
+ repo: string,
1528
+ maxContinuations: number,
1529
+ mode: string,
1530
+ model?: string,
1531
+ maxIterations?: number,
1532
+ ): ContinuationDecision {
1533
+ const current = typeof group.continuationCount === "number" ? group.continuationCount : 0
1534
+ if (group.exit_code === 0 || !(isIterationExhaustion(group.summary) || isProviderFailure(group.summary))) {
1535
+ // Real failure — budget stops and every other exit reason stay
1536
+ // terminal. No patch: the watcher's "failed" stands.
1537
+ return { shouldSpawn: false, newContinuationCount: current, patch: {} }
1538
+ }
1539
+ const newContinuationCount = current + 1
1540
+ const now = new Date().toISOString()
1541
+
1542
+ if (current < maxContinuations) {
1543
+ const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-continue${newContinuationCount}.md`)
1544
+ const worktreePath = groupWorktreePath(repo, group)
1545
+ // The session that just hit the cap wrote its own condensed history
1546
+ // here (src/engine/handoff.ts) right before exiting — inline it into
1547
+ // the continuation task file so the new conversation doesn't have to
1548
+ // re-derive everything from scratch, then clear it: it's about to be
1549
+ // baked into the task file text, and a stale copy must never leak
1550
+ // into a LATER, unrelated continuation on this same worktree.
1551
+ const handoffSummary = readHandoffSummary(worktreePath)
1552
+ clearHandoffSummary(worktreePath)
1553
+ const taskContent = buildContinuationTaskFileContent(group, newContinuationCount, group.issueBodies, handoffSummary)
1554
+ const spawnCommand =
1555
+ `bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
1556
+ `"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
1557
+ (model ? ` --model "${model}"` : "") +
1558
+ (maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
1559
+ return {
1560
+ shouldSpawn: true,
1561
+ newContinuationCount,
1562
+ taskFilePath,
1563
+ taskContent,
1564
+ spawnCommand,
1565
+ patch: {
1566
+ status: "running",
1567
+ continuationCount: newContinuationCount,
1568
+ // Fresh spawned timestamp: the stall guard must measure the
1569
+ // continuation worker, not the (possibly hours-old) attempt
1570
+ // that hit the cap.
1571
+ spawned: now,
1572
+ // A fresh session starts clean: drop the previous attempt's
1573
+ // completion artifacts, review/QA records and stall flag.
1574
+ review_verdict: undefined,
1575
+ pending_review_findings: undefined,
1576
+ reviewed_at: undefined,
1577
+ review_report: undefined,
1578
+ qa: undefined,
1579
+ stalled: undefined,
1580
+ exit_code: undefined,
1581
+ summary: undefined,
1582
+ // Must also clear cost_recorded — see the identical note in
1583
+ // handleReviewVerdict's patch above.
1584
+ cost_recorded: undefined,
1585
+ last_activity: {
1586
+ note: `continuation ${newContinuationCount}: previous session hit the iteration cap; re-spawned worker on the same worktree`,
1587
+ },
1588
+ },
1589
+ }
1590
+ }
1591
+
1592
+ // Cap reached: no new attempt. Terminal "needs-human" outcome.
1593
+ return {
1594
+ shouldSpawn: false,
1595
+ newContinuationCount: current,
1596
+ patch: {
1597
+ status: "needs-human",
1598
+ continuationCount: current,
1599
+ last_activity: {
1600
+ note: `exhausted ${current} continuation attempt(s) — still hitting the iteration cap; needs human`,
1601
+ },
1602
+ },
1603
+ }
1604
+ }
1605
+
1606
+ // ─── Spawn command assembly ──────────────────────────────────────────────────
1607
+
1608
+ function spawnCommandFor(harnessRoot: string, specs: WorktreeSpec[]): string {
1609
+ const triples = specs.map((spec) => `${spec.name}:${specs.indexOf(spec)}:plans/parallel-tasks/${spec.taskFile}`)
1610
+ return `bash ${path.join(harnessRoot, "scripts", "spawn-parallel-worktrees.sh")} ${triples.join(" ")}`
1611
+ }
1612
+
1613
+ export interface BuildSpawnEnvOptions {
1614
+ repo: string
1615
+ /** Harness mode for the workers (e.g. "code"). */
1616
+ mode: string
1617
+ /** An explicit --model flag value, if the caller passed one — always wins (resolveModelForMode rule 1). */
1618
+ explicitModel?: string
1619
+ memoryDir?: string
1620
+ /** Per-session iteration cap forwarded as HEADLESSCODE_MAX_ITERATIONS (run-worker.sh → --max-iterations). */
1621
+ maxIterations?: number
1622
+ /** Issue #49: run a plan-first session per worktree before the code worker (forwarded as PLAN_FIRST=1). */
1623
+ planFirst?: boolean
1624
+ /** Mode slug for the plan-first session (default: architect). */
1625
+ planFirstMode?: string
1626
+ /** Iteration cap for the plan-first session (default: 15). */
1627
+ planFirstMaxIterations?: number
1628
+ /** Env to read OPENROUTER_MODEL from (default: process.env). */
1629
+ env?: NodeJS.ProcessEnv
1630
+ }
1631
+
1632
+ /**
1633
+ * Resolve the worker's model and build the env for the initial spawnSync
1634
+ * (step 2 of orchestrateMain). Exported for tests: this is the fix for
1635
+ * "mode-models.json ignored on a worker's FIRST spawn" — the resolved model
1636
+ * is threaded into the spawn env as OPENROUTER_MODEL (which the spawner
1637
+ * script and run-worker.sh inherit), instead of only applying on rework
1638
+ * re-spawns. The returned workerModel is the SAME const the rework loop
1639
+ * reuses later.
1640
+ */
1641
+ export function buildSpawnEnv(options: BuildSpawnEnvOptions): { env: NodeJS.ProcessEnv; workerModel: string | undefined } {
1642
+ const workerModel = resolveModelForMode({
1643
+ workspaceRoot: options.repo,
1644
+ mode: options.mode,
1645
+ explicitModel: options.explicitModel,
1646
+ env: options.env ?? process.env,
1647
+ })
1648
+ const env: NodeJS.ProcessEnv = {
1649
+ ...(options.env ?? process.env),
1650
+ TARGET_REPO: options.repo,
1651
+ ORCHESTRATOR_MODE: options.mode,
1652
+ // The resolved worker model must reach the spawner (and the workers it
1653
+ // launches) — same pattern as HEADLESSCODE_PROJECT below.
1654
+ ...(workerModel ? { OPENROUTER_MODEL: workerModel } : {}),
1655
+ // Phase 3: workers scope memory to the REPO (not the worktree) and
1656
+ // share one memory dir. HEADLESSCODE_PROJECT flows to the worker CLI
1657
+ // which passes it as the session `project`; the env var is inherited
1658
+ // when --memory-dir is not given.
1659
+ HEADLESSCODE_PROJECT: path.basename(options.repo),
1660
+ ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
1661
+ // Iteration cap: an explicit --max-iterations beats the inherited env
1662
+ // var; undefined leaves whatever the caller's env already had in place
1663
+ // (or the harness's own default, 50, if neither is set).
1664
+ ...(options.maxIterations !== undefined ? { HEADLESSCODE_MAX_ITERATIONS: String(options.maxIterations) } : {}),
1665
+ // Operator knobs (spawner guardrail + local exploration) forwarded from
1666
+ // the AMBIENT env, not options.env — the caller's env is usually
1667
+ // process.env so the spread above already carries them, but a filtered
1668
+ // env must not silently drop an operator decision. ALLOW_UNINDEXED /
1669
+ // HEADLESSCODE_AUTO_INDEX gate the spawner's no-index guardrail;
1670
+ // HEADLESSCODE_LOCAL_EXPLORE* enable the pre-cloud local-explore phase.
1671
+ ...(process.env.ALLOW_UNINDEXED ? { ALLOW_UNINDEXED: process.env.ALLOW_UNINDEXED } : {}),
1672
+ ...(process.env.HEADLESSCODE_AUTO_INDEX ? { HEADLESSCODE_AUTO_INDEX: process.env.HEADLESSCODE_AUTO_INDEX } : {}),
1673
+ ...(process.env.HEADLESSCODE_LOCAL_EXPLORE ? { HEADLESSCODE_LOCAL_EXPLORE: process.env.HEADLESSCODE_LOCAL_EXPLORE } : {}),
1674
+ ...(process.env.HEADLESSCODE_LOCAL_EXPLORE_MODEL
1675
+ ? { HEADLESSCODE_LOCAL_EXPLORE_MODEL: process.env.HEADLESSCODE_LOCAL_EXPLORE_MODEL }
1676
+ : {}),
1677
+ ...(process.env.HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS
1678
+ ? { HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS: process.env.HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS }
1679
+ : {}),
1680
+ ...(process.env.HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS
1681
+ ? { HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS: process.env.HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS }
1682
+ : {}),
1683
+ // Issue #49 plan-first experiment: forwarded to the spawner, which runs
1684
+ // a short architect-mode planning session in each worktree BEFORE
1685
+ // launching the code worker and appends the plan (PLAN.md) to the
1686
+ // worker's task file. Only set when --plan-first is on — never the
1687
+ // default. PLAN_FIRST_MODE/MAX_ITERATIONS are only meaningful when
1688
+ // PLAN_FIRST is set, so they travel together.
1689
+ ...(options.planFirst ? { PLAN_FIRST: "1" } : {}),
1690
+ ...(options.planFirst ? { PLAN_FIRST_MODE: options.planFirstMode ?? "architect" } : {}),
1691
+ ...(options.planFirst ? { PLAN_FIRST_MAX_ITERATIONS: String(options.planFirstMaxIterations ?? 15) } : {}),
1692
+ }
1693
+ return { env, workerModel }
1694
+ }
1695
+
1696
+ /**
1697
+ * Issue #53 pre-flight issue-size check: one loud stderr line per issue whose
1698
+ * body reads like 3+ independent pieces of work. Warning-only by design — a
1699
+ * crude heuristic can false-positive, so it never aborts the round; the
1700
+ * operator is told BEFORE anything spawns so they can split the issue first.
1701
+ */
1702
+ function printIssueSizeWarnings(warnings: IssueSizeWarning[]): void {
1703
+ for (const w of warnings) {
1704
+ process.stderr.write(
1705
+ `headlesscode orchestrate: WARNING: issue #${w.number} ("${w.title}") reads like ${w.sections} independent pieces of work ` +
1706
+ `(${w.sections} top-level numbered/bulleted sections in its body). Dispatching it to ONE worker risks iteration-cap / budget ` +
1707
+ `burn + rework cycles (issue #53). Consider splitting it into ${w.sections} sub-issues and re-running before spawning. ` +
1708
+ `--no-issue-size-check silences this warning.\n`,
1709
+ )
1710
+ }
1711
+ }
1712
+
1713
+ function printPlan(
1714
+ repo: string,
1715
+ mode: string,
1716
+ specs: WorktreeSpec[],
1717
+ issues: SplitIssue[],
1718
+ models?: { worker: string; reviewer: string; qa: string },
1719
+ maxIterations?: number,
1720
+ estimateSection?: string[],
1721
+ /** Issue #49: plan-first session info to surface on the dry-run plan. */
1722
+ planFirst?: { mode: string; maxIterations: number },
1723
+ ): void {
1724
+ // Shape labels come from split.ts's issueShape (the SAME taxonomy the
1725
+ // split and the cost estimate use) — a single source of truth instead of
1726
+ // a third inline copy of the keyword scans.
1727
+ const shapeNotes = new Map<number, string>()
1728
+ for (const issue of issues) {
1729
+ const shape = issueShape(issue)
1730
+ const label =
1731
+ shape === "hot" ? "hot-path isolated" : shape === "generic" ? "generic" : `same-shape: ${shape}`
1732
+ shapeNotes.set(issue.number, label)
1733
+ }
1734
+
1735
+ process.stdout.write("── Orchestrate plan (dry run) ─────────────────────────────\n")
1736
+ process.stdout.write(` repo: ${repo}\n`)
1737
+ process.stdout.write(` mode: ${mode}\n`)
1738
+ if (planFirst) {
1739
+ process.stdout.write(
1740
+ ` plan-first: mode ${planFirst.mode}, max ${planFirst.maxIterations} iterations per worktree ` +
1741
+ `(issue #49 — plan session runs BEFORE each code worker)\n`,
1742
+ )
1743
+ }
1744
+ process.stdout.write(
1745
+ maxIterations !== undefined
1746
+ ? ` max iterations: ${maxIterations} (workers get --max-iterations ${maxIterations})\n`
1747
+ : " max iterations: <harness default 250>\n",
1748
+ )
1749
+ if (models) {
1750
+ process.stdout.write(` models: worker: ${models.worker}\n`)
1751
+ process.stdout.write(` reviewer: ${models.reviewer}\n`)
1752
+ process.stdout.write(` QA: ${models.qa}\n`)
1753
+ }
1754
+ process.stdout.write(` issues: ${issues.map((i) => i.number).join(", ")}\n\n`)
1755
+ process.stdout.write("Split plan (heuristics from the multi-agent-orchestrator mode):\n")
1756
+ for (const spec of specs) {
1757
+ const reasons = spec.issues.map((n) => shapeNotes.get(n) ?? "generic").join(", ")
1758
+ process.stdout.write(
1759
+ ` ${spec.name.padEnd(4)} issue ${spec.issues.join(", ").padEnd(10)} -> plans/parallel-tasks/${spec.taskFile} (${reasons})\n`,
1760
+ )
1761
+ }
1762
+ // Issue #16: expected cost/iteration range for this round, estimated
1763
+ // from recorded cost history (best-effort — a read failure yields no
1764
+ // section, never a failed dry-run).
1765
+ if (estimateSection && estimateSection.length > 0) {
1766
+ process.stdout.write("\nCost estimate (from recorded cost-history — issue #16):\n")
1767
+ for (const line of estimateSection) {
1768
+ process.stdout.write(` ${line}\n`)
1769
+ }
1770
+ }
1771
+ process.stdout.write("\nSpawn commands:\n")
1772
+ process.stdout.write(` ${spawnCommandFor(HARNESS_ROOT_TS, specs)}\n`)
1773
+ }
1774
+
1775
+ /** Resolve this harness repo's root (parent of src/orchestrator). */
1776
+ const HARNESS_ROOT_TS = fileURLToPath(new URL("../..", import.meta.url))
1777
+
1778
+ // ─── `orchestrate status` — read-side status / wait for EXTERNAL callers ────
1779
+
1780
+ const STATUS_USAGE = `headlesscode orchestrate status — per-group status of a round, optionally waiting for it
1781
+
1782
+ Usage:
1783
+ headlesscode orchestrate status --repo <path> [--json]
1784
+ headlesscode orchestrate status --repo <path> --wait [--timeout-ms <n>] [--on-group-terminal <cmd>] [--json]
1785
+
1786
+ Options:
1787
+ --repo <path> Target repo root whose .worktrees/.orchestrator-state.json
1788
+ is read (required). Before reporting, stale non-terminal
1789
+ entries are reconciled against real worktree markers
1790
+ (.harness.done/.harness.exit) and the state file is
1791
+ patched back when ground truth proves a different status
1792
+ ("done"/"failed", or "orphaned" when the worktree is gone).
1793
+ --wait Block inside this ONE call until every group reaches a
1794
+ terminal status (done|failed|needs-human|orphaned;
1795
+ "blocked" is NOT terminal — it keeps waiting) or
1796
+ --timeout-ms elapses, then print the same summary plus a
1797
+ one-line verdict. Reconciliation runs on every poll.
1798
+ --timeout-ms <n> Max --wait duration, ms (default: 7200000 = 2h, matching
1799
+ the watcher's stall guard — a round left non-terminal
1800
+ longer than that is flagged stalled anyway).
1801
+ --poll-interval-ms <n> State-file poll interval for --wait, ms (default: 5000,
1802
+ matching the watcher's own write cadence).
1803
+ --on-group-terminal <cmd>
1804
+ With --wait: run <cmd> (spawned directly — no shell
1805
+ parsing — with the group name and its terminal status as
1806
+ two positional args) each time a group transitions to
1807
+ done|failed|needs-human|orphaned mid-wait. Fires once per
1808
+ group per transition; never fires for groups already
1809
+ terminal when the wait started. A failing hook logs a
1810
+ warning to stderr but never aborts the wait.
1811
+ --json Machine-readable output. One-shot: the reconciled
1812
+ OrchestratorState plus a top-level "reconciled" array of
1813
+ group names patched (only when something was patched).
1814
+ With --wait: the final state plus top-level
1815
+ allDone/timedOut/verdict/reconciled fields.
1816
+ --help, -h Show this help and exit.
1817
+
1818
+ Exit codes:
1819
+ 0 one-shot: no group is failed/needs-human/orphaned right now;
1820
+ --wait: every group is done (clean finish)
1821
+ 1 one-shot: at least one group is failed, needs-human, or orphaned;
1822
+ --wait: any group failed/needs-human/orphaned, or the wait timed out
1823
+ 2 usage error (bad or missing arguments)
1824
+ `
1825
+
1826
+ export interface StatusOptions {
1827
+ repo: string
1828
+ json: boolean
1829
+ wait: boolean
1830
+ timeoutMs: number
1831
+ pollIntervalMs: number
1832
+ /** Command spawned (with the group name + terminal status as args) per terminal transition while --wait is active. */
1833
+ onGroupTerminal?: string
1834
+ help: boolean
1835
+ }
1836
+
1837
+ export function parseStatusArgs(argv: string[]): { options: StatusOptions; error?: string } {
1838
+ const options: StatusOptions = {
1839
+ repo: "",
1840
+ json: false,
1841
+ wait: false,
1842
+ timeoutMs: DEFAULT_STATUS_TIMEOUT_MS,
1843
+ pollIntervalMs: DEFAULT_STATUS_POLL_INTERVAL_MS,
1844
+ onGroupTerminal: undefined,
1845
+ help: false,
1846
+ }
1847
+ for (let i = 0; i < argv.length; i++) {
1848
+ const arg = argv[i]
1849
+ const eq = arg.indexOf("=")
1850
+ const flag = eq === -1 ? arg : arg.slice(0, eq)
1851
+ const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
1852
+ const next = (): string | undefined => {
1853
+ if (inlineValue !== undefined) {
1854
+ return inlineValue
1855
+ }
1856
+ const v = argv[i + 1]
1857
+ if (v === undefined || v.startsWith("--")) {
1858
+ return undefined
1859
+ }
1860
+ i++
1861
+ return v
1862
+ }
1863
+ switch (flag) {
1864
+ case "--repo": {
1865
+ const v = next()
1866
+ if (v === undefined) {
1867
+ return { options, error: "Missing value for --repo" }
1868
+ }
1869
+ options.repo = v
1870
+ break
1871
+ }
1872
+ case "--json":
1873
+ options.json = true
1874
+ break
1875
+ case "--wait":
1876
+ options.wait = true
1877
+ break
1878
+ case "--timeout-ms": {
1879
+ const v = next()
1880
+ const n = v === undefined ? Number.NaN : Number(v)
1881
+ if (!Number.isInteger(n) || n <= 0) {
1882
+ return { options, error: "--timeout-ms requires a positive integer" }
1883
+ }
1884
+ options.timeoutMs = n
1885
+ break
1886
+ }
1887
+ case "--poll-interval-ms": {
1888
+ const v = next()
1889
+ const n = v === undefined ? Number.NaN : Number(v)
1890
+ if (!Number.isInteger(n) || n <= 0) {
1891
+ return { options, error: "--poll-interval-ms requires a positive integer" }
1892
+ }
1893
+ options.pollIntervalMs = n
1894
+ break
1895
+ }
1896
+ case "--on-group-terminal": {
1897
+ const v = next()
1898
+ if (v === undefined) {
1899
+ return { options, error: "Missing value for --on-group-terminal" }
1900
+ }
1901
+ options.onGroupTerminal = v
1902
+ break
1903
+ }
1904
+ case "--help":
1905
+ case "-h":
1906
+ options.help = true
1907
+ break
1908
+ default:
1909
+ return { options, error: `Unknown orchestrate status argument: ${arg}` }
1910
+ }
1911
+ }
1912
+ return { options }
1913
+ }
1914
+
1915
+ export interface StatusIo {
1916
+ stdout?: (text: string) => void
1917
+ stderr?: (text: string) => void
1918
+ }
1919
+
1920
+ function waitExitCode(allDone: boolean, summary: StatusSummary): number {
1921
+ // Nothing in the state file is not a failure — exit 0 so a caller that
1922
+ // ran --wait against a wrong/empty path sees a clear, non-alarming result.
1923
+ if (summary.counts.total === 0) {
1924
+ return 0
1925
+ }
1926
+ if (allDone) {
1927
+ return 0
1928
+ }
1929
+ // Reached here only when allDone is false: either the wait timed out with
1930
+ // groups still non-terminal, or every group is terminal but at least one
1931
+ // failed/needs-human. Both are non-clean outcomes → 1.
1932
+ return 1
1933
+ }
1934
+
1935
+ /**
1936
+ * Run the `--on-group-terminal <cmd>` hook for a group that just reached a
1937
+ * terminal status: spawn <cmd> directly (NO shell parsing of <cmd> — it is
1938
+ * the executable path/name) with the group name and status as the two
1939
+ * positional args, and await completion so transitions fire sequentially.
1940
+ * A broken hook (spawn failure or non-zero exit) logs a warning to stderr
1941
+ * but never aborts the wait — a failed notification must not take down a
1942
+ * status watch that is otherwise healthy.
1943
+ */
1944
+ async function runGroupTerminalHook(cmd: string, group: OrchestratorGroup, writeErr: (text: string) => void): Promise<void> {
1945
+ try {
1946
+ const child = spawn(cmd, [group.name, group.status], { stdio: "inherit" })
1947
+ await new Promise<void>((resolve, reject) => {
1948
+ child.once("error", reject)
1949
+ child.once("exit", (code, signal) => {
1950
+ if (code !== 0) {
1951
+ reject(new Error(signal ? `killed by ${signal}` : `exit ${code}`))
1952
+ } else {
1953
+ resolve()
1954
+ }
1955
+ })
1956
+ })
1957
+ } catch (err) {
1958
+ writeErr(
1959
+ `headlesscode orchestrate status: --on-group-terminal hook failed for group "${group.name}" ` +
1960
+ `(${group.status}): ${err instanceof Error ? err.message : String(err)}\n`,
1961
+ )
1962
+ }
1963
+ }
1964
+
1965
+ /**
1966
+ * Attach the cleanup-eligibility column to every TERMINAL group's status row
1967
+ * (read-only, no side effects — the same gates the cleanup command uses,
1968
+ * minus the GitHub PR network check). Returns the per-group map for --json.
1969
+ * Never throws: on any failure the group is reported blocked with a reason,
1970
+ * matching cleanup's fail-closed model.
1971
+ */
1972
+ function attachCleanupStatus(repo: string, state: OrchestratorState, summary: StatusSummary): Record<string, CleanupStatus> {
1973
+ let baseBranch: string | undefined
1974
+ try {
1975
+ baseBranch = resolveBaseBranch(repo)
1976
+ } catch {
1977
+ baseBranch = undefined
1978
+ }
1979
+ const byGroup: Record<string, CleanupStatus> = {}
1980
+ for (const row of summary.groups) {
1981
+ if (!isTerminalStatus(row.status)) {
1982
+ continue // non-terminal groups are never touched by cleanup
1983
+ }
1984
+ const group = state.groups.find((g) => g.name === row.name)
1985
+ if (!group) {
1986
+ continue
1987
+ }
1988
+ const cleanup: CleanupStatus =
1989
+ baseBranch === undefined
1990
+ ? { status: "blocked", reason: "cannot resolve base branch (detached HEAD?) — pass --base to cleanup" }
1991
+ : assessGroupCleanupSync(repo, group, baseBranch)
1992
+ row.cleanup = cleanup
1993
+ byGroup[row.name] = cleanup
1994
+ }
1995
+ return byGroup
1996
+ }
1997
+
1998
+ /**
1999
+ * `headlesscode orchestrate status` entry point. One call, one result: the
2000
+ * calling agent never re-invokes a tool per poll, unlike the bash loop this
2001
+ * replaces. Both forms reconcile stale state against real worktree markers
2002
+ * before reporting (see status.ts's reconcileGroups) — the state file is
2003
+ * patched back when ground truth proves a different status, so a stale
2004
+ * "running" entry is never reported as-is.
2005
+ */
2006
+ export async function statusMain(argv: string[], io: StatusIo = {}): Promise<number> {
2007
+ const writeOut = io.stdout ?? ((text: string) => process.stdout.write(text))
2008
+ const writeErr = io.stderr ?? ((text: string) => process.stderr.write(text))
2009
+
2010
+ const { options, error } = parseStatusArgs(argv)
2011
+ if (error) {
2012
+ writeErr(`headlesscode orchestrate status: ${error}\n\n${STATUS_USAGE}`)
2013
+ return 2
2014
+ }
2015
+ if (options.help) {
2016
+ writeOut(STATUS_USAGE)
2017
+ return 0
2018
+ }
2019
+ if (!options.repo) {
2020
+ writeErr(`headlesscode orchestrate status: --repo <path> is required\n\n${STATUS_USAGE}`)
2021
+ return 2
2022
+ }
2023
+
2024
+ const repo = path.resolve(options.repo)
2025
+ const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
2026
+
2027
+ if (options.wait) {
2028
+ // --on-group-terminal: spawn the hook command per genuine terminal
2029
+ // transition (sequentially — terminal transitions are rare and
2030
+ // blocking the poll loop for the hook is fine).
2031
+ const hookCmd = options.onGroupTerminal
2032
+ const result = await waitForTerminalState(statePath, {
2033
+ timeoutMs: options.timeoutMs,
2034
+ pollIntervalMs: options.pollIntervalMs,
2035
+ // Reconcile on EVERY poll (a group can finish mid-wait). The
2036
+ // hook reports the patch; waitForTerminalState persists it ONLY
2037
+ // when no live watcher owns the state file (issue #116: a
2038
+ // concurrent `status --wait` persisting its read-side reconcile
2039
+ // pre-empted the watcher, which then short-circuited the group
2040
+ // and silently skipped log analysis + the automated review —
2041
+ // see waitForTerminalState's live-watcher heartbeat check).
2042
+ reconcile: (state) => reconcileGroups(state, repo),
2043
+ onGroupTerminal: hookCmd ? (group) => runGroupTerminalHook(hookCmd, group, writeErr) : undefined,
2044
+ })
2045
+ const summary = buildStatusSummary(result.state)
2046
+ const verdict = verdictLine(summary, result.timedOut, result.allDone)
2047
+ const cleanupByGroup = attachCleanupStatus(repo, result.state, summary)
2048
+ if (options.json) {
2049
+ const out: Record<string, unknown> = {
2050
+ ...result.state,
2051
+ allDone: result.allDone,
2052
+ timedOut: result.timedOut,
2053
+ verdict,
2054
+ }
2055
+ if (result.reconciled.length > 0) {
2056
+ out.reconciled = result.reconciled
2057
+ }
2058
+ if (Object.keys(cleanupByGroup).length > 0) {
2059
+ out.cleanup = cleanupByGroup
2060
+ }
2061
+ writeOut(JSON.stringify(out, null, 2) + "\n")
2062
+ } else {
2063
+ writeOut(
2064
+ formatStatusText(summary, {
2065
+ repo,
2066
+ statePath,
2067
+ timedOut: result.timedOut,
2068
+ allDone: result.allDone,
2069
+ elapsedMs: result.elapsedMs,
2070
+ reconciled: result.reconciled,
2071
+ }),
2072
+ )
2073
+ }
2074
+ return waitExitCode(result.allDone, summary)
2075
+ }
2076
+
2077
+ // One-shot: read the current state, reconcile it against real markers
2078
+ // (persisting any patch), print, exit on current health.
2079
+ let state: OrchestratorState
2080
+ try {
2081
+ state = loadStateSync(statePath)
2082
+ } catch (err) {
2083
+ writeErr(
2084
+ `headlesscode orchestrate status: cannot read state file ${statePath}: ${err instanceof Error ? err.message : String(err)}\n`,
2085
+ )
2086
+ return 1
2087
+ }
2088
+ const reconciliation = reconcileGroups(state, repo)
2089
+ if (reconciliation.reconciled.length > 0) {
2090
+ saveStateSync(statePath, reconciliation.state)
2091
+ state = reconciliation.state
2092
+ }
2093
+ const summary = buildStatusSummary(state)
2094
+ const cleanupByGroup = attachCleanupStatus(repo, state, summary)
2095
+ if (options.json) {
2096
+ const out: Record<string, unknown> = { ...state }
2097
+ if (reconciliation.reconciled.length > 0) {
2098
+ out.reconciled = reconciliation.reconciled
2099
+ }
2100
+ if (Object.keys(cleanupByGroup).length > 0) {
2101
+ out.cleanup = cleanupByGroup
2102
+ }
2103
+ writeOut(JSON.stringify(out, null, 2) + "\n")
2104
+ } else {
2105
+ writeOut(formatStatusText(summary, { repo, statePath, reconciled: reconciliation.reconciled }))
2106
+ }
2107
+ return summary.counts.failed > 0 || summary.counts.needsHuman > 0 || summary.counts.orphaned > 0 ? 1 : 0
2108
+ }
2109
+
2110
+ // ─── `orchestrate stop` — stop a group's WHOLE worker tree (issue #20) ──────
2111
+
2112
+ const STOP_USAGE = `headlesscode orchestrate stop — stop one or more groups' worker process trees
2113
+
2114
+ Usage:
2115
+ headlesscode orchestrate stop --repo <path> --group <name> [--group <name> ...] [options]
2116
+
2117
+ Stops a worker COMPLETELY: scripts/stop-worker.sh targets the worker's process
2118
+ GROUP — run-worker.sh launches each worker under setsid and the wrapper
2119
+ records its process-group id in <worktree>/.harness.pgid — so the wrapper
2120
+ bash, npx, node/tsx and every non-detached grandchild are all killed.
2121
+ Killing the wrapper PID alone used to leave the real work running undetected,
2122
+ reparented to init, until the session finished on its own (issue #20).
2123
+
2124
+ Each stopped group is patched to "needs-human" (a TERMINAL state — a human
2125
+ must decide what to do with the stopped worktree next, e.g. resume it under
2126
+ new logic). Groups that are already terminal are left untouched.
2127
+
2128
+ Options:
2129
+ --repo <path> Target repo root whose .worktrees/.orchestrator-state.json
2130
+ is read (required).
2131
+ --group <name> Worktree group name to stop (repeatable; at least one).
2132
+ --grace-ms <n> SIGTERM grace period before SIGKILL escalation, ms
2133
+ (default: 5000; forwarded to scripts/stop-worker.sh).
2134
+ --json Machine-readable output: per-group results + final state.
2135
+ --help, -h Show this help and exit.
2136
+
2137
+ Exit codes:
2138
+ 0 every requested group was stopped (or was already terminal)
2139
+ 1 a requested group could not be stopped / is not in the state file
2140
+ 2 usage error (bad or missing arguments)
2141
+ `
2142
+
2143
+ export interface StopOptions {
2144
+ repo: string
2145
+ groups: string[]
2146
+ graceMs: number
2147
+ json: boolean
2148
+ help: boolean
2149
+ }
2150
+
2151
+ export function parseStopArgs(argv: string[]): { options: StopOptions; error?: string } {
2152
+ const options: StopOptions = { repo: "", groups: [], graceMs: 5000, json: false, help: false }
2153
+ for (let i = 0; i < argv.length; i++) {
2154
+ const arg = argv[i]
2155
+ const eq = arg.indexOf("=")
2156
+ const flag = eq === -1 ? arg : arg.slice(0, eq)
2157
+ const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
2158
+ const next = (): string | undefined => {
2159
+ if (inlineValue !== undefined) {
2160
+ return inlineValue
2161
+ }
2162
+ const v = argv[i + 1]
2163
+ if (v === undefined || v.startsWith("--")) {
2164
+ return undefined
2165
+ }
2166
+ i++
2167
+ return v
2168
+ }
2169
+ switch (flag) {
2170
+ case "--repo": {
2171
+ const v = next()
2172
+ if (v === undefined) {
2173
+ return { options, error: "Missing value for --repo" }
2174
+ }
2175
+ options.repo = v
2176
+ break
2177
+ }
2178
+ case "--group": {
2179
+ const v = next()
2180
+ if (v === undefined) {
2181
+ return { options, error: "Missing value for --group" }
2182
+ }
2183
+ options.groups.push(v)
2184
+ break
2185
+ }
2186
+ case "--grace-ms": {
2187
+ const v = next()
2188
+ const n = v === undefined ? Number.NaN : Number(v)
2189
+ if (!Number.isInteger(n) || n < 0) {
2190
+ return { options, error: "--grace-ms requires a non-negative integer" }
2191
+ }
2192
+ options.graceMs = n
2193
+ break
2194
+ }
2195
+ case "--json":
2196
+ options.json = true
2197
+ break
2198
+ case "--help":
2199
+ case "-h":
2200
+ options.help = true
2201
+ break
2202
+ default:
2203
+ return { options, error: `Unknown orchestrate stop argument: ${arg}` }
2204
+ }
2205
+ }
2206
+ return { options }
2207
+ }
2208
+
2209
+ export interface StopIo {
2210
+ stdout?: (text: string) => void
2211
+ stderr?: (text: string) => void
2212
+ /** Injectable stop-script runner (tests); default: spawnSync scripts/stop-worker.sh. */
2213
+ runStopWorker?: (wtPath: string, graceMs: number) => { exitCode: number; output: string }
2214
+ }
2215
+
2216
+ /**
2217
+ * `headlesscode orchestrate stop` entry point. Stops each requested group's
2218
+ * worker process tree via scripts/stop-worker.sh (issue #20) and patches the
2219
+ * stopped groups to `needs-human` so they never sit "running" forever with no
2220
+ * process behind them.
2221
+ */
2222
+ export async function stopMain(argv: string[], io: StopIo = {}): Promise<number> {
2223
+ const writeOut = io.stdout ?? ((text: string) => process.stdout.write(text))
2224
+ const writeErr = io.stderr ?? ((text: string) => process.stderr.write(text))
2225
+
2226
+ const { options, error } = parseStopArgs(argv)
2227
+ if (error) {
2228
+ writeErr(`headlesscode orchestrate stop: ${error}\n\n${STOP_USAGE}`)
2229
+ return 2
2230
+ }
2231
+ if (options.help) {
2232
+ writeOut(STOP_USAGE)
2233
+ return 0
2234
+ }
2235
+ if (!options.repo) {
2236
+ writeErr(`headlesscode orchestrate stop: --repo <path> is required\n\n${STOP_USAGE}`)
2237
+ return 2
2238
+ }
2239
+ if (options.groups.length === 0) {
2240
+ writeErr(`headlesscode orchestrate stop: at least one --group <name> is required\n\n${STOP_USAGE}`)
2241
+ return 2
2242
+ }
2243
+
2244
+ const repo = path.resolve(options.repo)
2245
+ const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
2246
+ const runStopWorker =
2247
+ io.runStopWorker ??
2248
+ ((wtPath: string, graceMs: number): { exitCode: number; output: string } => {
2249
+ const res = spawnSync(
2250
+ "bash",
2251
+ [path.join(HARNESS_ROOT_TS, "scripts", "stop-worker.sh"), wtPath, "--grace-ms", String(graceMs)],
2252
+ { encoding: "utf-8" },
2253
+ )
2254
+ const output = `${res.stdout ?? ""}${res.stderr ?? ""}`.trim()
2255
+ return { exitCode: res.status ?? 1, output }
2256
+ })
2257
+
2258
+ if (!fs.existsSync(statePath)) {
2259
+ writeErr(`headlesscode orchestrate stop: no orchestrator state file at ${statePath} — nothing to stop (wrong --repo?)\n`)
2260
+ return 1
2261
+ }
2262
+
2263
+ let state: OrchestratorState
2264
+ try {
2265
+ state = loadStateSync(statePath)
2266
+ } catch (err) {
2267
+ writeErr(
2268
+ `headlesscode orchestrate stop: cannot read state file ${statePath}: ${err instanceof Error ? err.message : String(err)}\n`,
2269
+ )
2270
+ return 1
2271
+ }
2272
+
2273
+ // Validate EVERY requested group exists BEFORE stopping anything — an
2274
+ // operator typo must not stop half the requested set.
2275
+ const unknown = options.groups.filter((name) => !state.groups.some((g) => g.name === name))
2276
+ if (unknown.length > 0) {
2277
+ writeErr(
2278
+ `headlesscode orchestrate stop: group(s) not in state file ${statePath}: ${unknown.join(", ")}\n`,
2279
+ )
2280
+ return 1
2281
+ }
2282
+
2283
+ const results: Array<{ name: string; status: string; stopped: boolean; output?: string }> = []
2284
+ let failed = 0
2285
+ let current = state
2286
+ for (const name of options.groups) {
2287
+ const group = current.groups.find((g) => g.name === name)
2288
+ if (!group) {
2289
+ continue
2290
+ }
2291
+ // With --json, stdout stays pure JSON — human progress + the stop
2292
+ // script's own output go to stderr (conventional diagnostics channel).
2293
+ const writeDiag = options.json ? writeErr : writeOut
2294
+ if (isTerminalStatus(group.status)) {
2295
+ // Already done/failed/needs-human/orphaned: no live worker to stop,
2296
+ // and patching it would clobber the real outcome.
2297
+ writeDiag(`[stop] ${name}: already ${group.status} — nothing to stop\n`)
2298
+ results.push({ name, status: group.status, stopped: false })
2299
+ continue
2300
+ }
2301
+ const wtPath = groupWorktreePath(repo, group)
2302
+ const res = runStopWorker(wtPath, options.graceMs)
2303
+ if (res.exitCode !== 0) {
2304
+ writeDiag(res.output ? `${res.output}\n` : "")
2305
+ writeErr(
2306
+ `[stop] ${name}: scripts/stop-worker.sh failed (exit ${res.exitCode}) — group left as-is (${group.status})\n`,
2307
+ )
2308
+ failed++
2309
+ results.push({ name, status: group.status, stopped: false, ...(res.output ? { output: res.output } : {}) })
2310
+ continue
2311
+ }
2312
+ current = updateGroup(current, name, {
2313
+ status: "needs-human",
2314
+ stopped_at: new Date().toISOString(),
2315
+ stalled: undefined,
2316
+ blocked: undefined,
2317
+ last_activity: {
2318
+ note: "stopped by operator: worker process tree killed (scripts/stop-worker.sh); a human must decide next steps",
2319
+ },
2320
+ })
2321
+ results.push({ name, status: "needs-human", stopped: true })
2322
+ writeDiag(res.output ? `${res.output}\n` : "")
2323
+ writeDiag(`[stop] ${name}: worker tree stopped; group marked needs-human\n`)
2324
+ }
2325
+ saveStateSync(statePath, current)
2326
+
2327
+ if (options.json) {
2328
+ const out: Record<string, unknown> = {
2329
+ results,
2330
+ stopped: results.filter((r) => r.stopped).map((r) => r.name),
2331
+ groups: current.groups,
2332
+ }
2333
+ writeOut(JSON.stringify(out, null, 2) + "\n")
2334
+ }
2335
+ return failed > 0 ? 1 : 0
2336
+ }
2337
+
2338
+ // ─── `pipeline` subcommand (issue #148) ───────────────────────────────────────
2339
+
2340
+ const PIPELINE_USAGE = `headlesscode pipeline — run the stage-isolated research→filing pipeline
2341
+
2342
+ Usage:
2343
+ headlesscode pipeline --workspace <path> [--research-task <text>] [--skip-filing]
2344
+ [--model <id>] [--max-iterations <n>]
2345
+
2346
+ Runs the stage-isolated pipeline from issue #148: a fresh \`researcher\`-mode
2347
+ session produces ONE well-cited research doc on disk (the artifact gate is
2348
+ bound by default), then a fresh \`issue-filer\`-mode session turns that doc's
2349
+ CONTENT (read off disk — never the prior session's conversation) into real
2350
+ GitHub issue(s). Each stage is a genuinely separate \`HeadlessSession\` with
2351
+ its own context, mode, and executor.
2352
+
2353
+ Options:
2354
+ --workspace <path> Workspace root the stages run against (required; must be
2355
+ a git repo — the filer needs \`gh\` against its origin)
2356
+ --research-task <t> Task text for the research stage (default: a built-in
2357
+ prompt naming the artifact pattern + required sections)
2358
+ --skip-filing Stop after the research stage; do NOT run the filing stage
2359
+ --model <id> Model override for both stages (default: env / client)
2360
+ --max-iterations <n> Per-stage iteration cap (default: 60)
2361
+ --help Show this help and exit
2362
+
2363
+ Exit codes:
2364
+ 0 all run stages succeeded
2365
+ 1 a run stage failed (research session error / no artifact / filing parse error)
2366
+ 2 usage error (missing --workspace, unknown flag)
2367
+ `
2368
+
2369
+ interface PipelineCliOptions {
2370
+ workspace?: string
2371
+ researchTask?: string
2372
+ skipFiling: boolean
2373
+ model?: string
2374
+ maxIterations?: number
2375
+ help: boolean
2376
+ }
2377
+
2378
+ export function parsePipelineArgs(argv: string[]): { options: PipelineCliOptions; error?: string } {
2379
+ const options: PipelineCliOptions = { skipFiling: false, help: false }
2380
+ for (let i = 0; i < argv.length; i++) {
2381
+ const arg = argv[i]
2382
+ const eq = arg.indexOf("=")
2383
+ const flag = eq === -1 ? arg : arg.slice(0, eq)
2384
+ const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
2385
+ const next = (): string | undefined => {
2386
+ if (inlineValue !== undefined) {
2387
+ return inlineValue
2388
+ }
2389
+ const v = argv[i + 1]
2390
+ if (v === undefined || v.startsWith("--")) {
2391
+ return undefined
2392
+ }
2393
+ i++
2394
+ return v
2395
+ }
2396
+ switch (flag) {
2397
+ case "--workspace":
2398
+ case "--research-task":
2399
+ case "--model":
2400
+ case "--max-iterations": {
2401
+ const value = next()
2402
+ if (value === undefined) {
2403
+ return { options, error: `Missing value for ${flag}` }
2404
+ }
2405
+ if (flag === "--workspace") {
2406
+ options.workspace = value
2407
+ } else if (flag === "--research-task") {
2408
+ options.researchTask = value
2409
+ } else if (flag === "--model") {
2410
+ options.model = value
2411
+ } else {
2412
+ const n = Number(value)
2413
+ if (!Number.isInteger(n) || n <= 0) {
2414
+ return { options, error: "--max-iterations requires a positive integer" }
2415
+ }
2416
+ options.maxIterations = n
2417
+ }
2418
+ break
2419
+ }
2420
+ case "--skip-filing":
2421
+ options.skipFiling = true
2422
+ break
2423
+ case "--help":
2424
+ case "-h":
2425
+ options.help = true
2426
+ break
2427
+ default:
2428
+ return { options, error: `Unknown argument: ${arg}` }
2429
+ }
2430
+ }
2431
+ return { options }
2432
+ }
2433
+
2434
+ /**
2435
+ * `headlesscode pipeline` — run the stage-isolated research→filing pipeline
2436
+ * (issue #148) against one workspace. Research produces the artifact; filing
2437
+ * (unless --skip-filing) turns it into real GitHub issues.
2438
+ */
2439
+ export async function pipelineMain(argv: string[]): Promise<number> {
2440
+ const { options, error } = parsePipelineArgs(argv)
2441
+ if (error) {
2442
+ process.stderr.write(`headlesscode pipeline: ${error}\n\n${PIPELINE_USAGE}`)
2443
+ return 2
2444
+ }
2445
+ if (options.help) {
2446
+ process.stdout.write(PIPELINE_USAGE)
2447
+ return 0
2448
+ }
2449
+ if (!options.workspace) {
2450
+ process.stderr.write(`headlesscode pipeline: --workspace <path> is required\n\n${PIPELINE_USAGE}`)
2451
+ return 2
2452
+ }
2453
+
2454
+ const workspaceRoot = path.resolve(options.workspace)
2455
+ process.stdout.write(`headlesscode pipeline: running research stage (mode researcher, artifact gate ON)...\n`)
2456
+ const research = await runResearchStage({
2457
+ workspaceRoot,
2458
+ taskText: options.researchTask,
2459
+ model: options.model,
2460
+ maxIterations: options.maxIterations,
2461
+ })
2462
+ if (research.status !== "ok" || !research.artifactPath) {
2463
+ process.stderr.write(
2464
+ `headlesscode pipeline: research stage failed — ${research.summary}\n`,
2465
+ )
2466
+ return 1
2467
+ }
2468
+ process.stdout.write(`headlesscode pipeline: research artifact at ${research.artifactPath}\n`)
2469
+ if (options.skipFiling) {
2470
+ return 0
2471
+ }
2472
+
2473
+ process.stdout.write(`headlesscode pipeline: running filing stage (mode issue-filer)...\n`)
2474
+ const filing = await runFilingStage({
2475
+ workspaceRoot,
2476
+ researchArtifactPath: research.artifactPath,
2477
+ model: options.model,
2478
+ maxIterations: options.maxIterations,
2479
+ })
2480
+ if (filing.status !== "ok") {
2481
+ process.stderr.write(
2482
+ `headlesscode pipeline: filing stage failed — ${filing.summary}\n`,
2483
+ )
2484
+ return 1
2485
+ }
2486
+ process.stdout.write(
2487
+ `headlesscode pipeline: filed issue(s): ${filing.issueNumbers.join(", ")}\n`,
2488
+ )
2489
+ return 0
2490
+ }
2491
+
2492
+ // ─── Main ────────────────────────────────────────────────────────────────────
2493
+
2494
+ export async function orchestrateMain(argv: string[]): Promise<number> {
2495
+ // Subcommand dispatch: `headlesscode orchestrate status ...` is the
2496
+ // read-side status/wait command, `... stop ...` stops a group's whole
2497
+ // worker tree (issue #20), `... cleanup ...` is the human-triggered
2498
+ // post-merge worktree cleanup, `... review/rework/resume ...` (issue #14)
2499
+ // are the standalone recovery subcommands; everything else is the
2500
+ // spawn+watch round.
2501
+ if (argv[0] === "status") {
2502
+ return statusMain(argv.slice(1))
2503
+ }
2504
+ if (argv[0] === "stop") {
2505
+ return stopMain(argv.slice(1))
2506
+ }
2507
+ if (argv[0] === "cleanup") {
2508
+ return cleanupMain(argv.slice(1))
2509
+ }
2510
+ if (argv[0] === "pipeline") {
2511
+ return pipelineMain(argv.slice(1))
2512
+ }
2513
+ // Issue #14 standalone recovery subcommands. Imported lazily (not
2514
+ // statically) because resume.ts imports this module's rework-decision
2515
+ // helpers — a static import here would create a module cycle.
2516
+ if (argv[0] === "review" || argv[0] === "rework" || argv[0] === "resume") {
2517
+ const { reviewMain, reworkMain, resumeMain } = await import("./resume.js")
2518
+ switch (argv[0]) {
2519
+ case "review":
2520
+ return reviewMain(argv.slice(1))
2521
+ case "rework":
2522
+ return reworkMain(argv.slice(1))
2523
+ default:
2524
+ return resumeMain(argv.slice(1))
2525
+ }
2526
+ }
2527
+ const { options, error } = parseOrchestrateArgs(argv)
2528
+ if (error) {
2529
+ process.stderr.write(`headlesscode orchestrate: ${error}\n\n${ORCHESTRATE_USAGE}`)
2530
+ return 2
2531
+ }
2532
+ if (options.help) {
2533
+ process.stdout.write(ORCHESTRATE_USAGE)
2534
+ return 0
2535
+ }
2536
+ if (!options.repo) {
2537
+ process.stderr.write(`headlesscode orchestrate: --repo <path> is required\n\n${ORCHESTRATE_USAGE}`)
2538
+ return 2
2539
+ }
2540
+
2541
+ const repo = path.resolve(options.repo)
2542
+ try {
2543
+ execFileSync("git", ["-C", repo, "rev-parse", "--git-dir"], { stdio: "ignore", timeout: 5000 })
2544
+ } catch {
2545
+ process.stderr.write(`headlesscode orchestrate: not a git repo: ${repo}\n`)
2546
+ return 2
2547
+ }
2548
+
2549
+ let issues: SplitIssue[]
2550
+ try {
2551
+ issues = loadIssues(options)
2552
+ } catch (err) {
2553
+ process.stderr.write(`headlesscode orchestrate: ${err instanceof Error ? err.message : String(err)}\n`)
2554
+ return 2
2555
+ }
2556
+
2557
+ // Fix 3: --file-issues (only valid with --issues-json) files a REAL GitHub
2558
+ // issue for every synthetic --issues-json entry and substitutes the real
2559
+ // number, so worktree/branch naming, task files, and PR-closing comments
2560
+ // all use a number that actually exists on GitHub. Runs before anything
2561
+ // else touches `issues`. A real, visible-to-GitHub side effect — never
2562
+ // silent: one confirmation line per created issue.
2563
+ if (options.fileIssues) {
2564
+ const ownerRepo = originOwnerRepo(repo)
2565
+ if (!ownerRepo) {
2566
+ process.stderr.write(
2567
+ `headlesscode orchestrate: --file-issues needs a parseable GitHub 'origin' remote on ${repo} ` +
2568
+ `(got none) to know where to file the issues\n`,
2569
+ )
2570
+ return 2
2571
+ }
2572
+ try {
2573
+ const { issues: filed, created } = fileSyntheticIssues(issues, (issue) => createGhIssue(repo, ownerRepo, issue))
2574
+ for (const c of created) {
2575
+ process.stdout.write(`[orchestrate] filed issue #${c.number}: ${c.title} -> ${c.url}\n`)
2576
+ }
2577
+ issues = filed
2578
+ } catch (err) {
2579
+ process.stderr.write(
2580
+ `headlesscode orchestrate: --file-issues failed: ${err instanceof Error ? err.message : String(err)}\n`,
2581
+ )
2582
+ return 2
2583
+ }
2584
+ }
2585
+
2586
+ // Issue #53 pre-flight issue-size check + auto-split: a FREE deterministic
2587
+ // scan (split.ts's topLevelSectionCount) flags issues whose body reads
2588
+ // like 3+ independent pieces of work. Runs BEFORE splitIssues() below —
2589
+ // unlike the original warn-only version, this can REPLACE a flagged issue
2590
+ // with real sub-issues, and splitIssues must see the replacement, not the
2591
+ // oversized original. Auto-split needs a real (non-synthetic) issue to
2592
+ // close and a GitHub remote to file into; --issues-json entries and
2593
+ // remote-less repos fall back to warn-only (auto-split has nothing to
2594
+ // close/file against). --dry-run ALSO falls back to warn-only: filing
2595
+ // real issues and closing the parent are exactly the kind of visible,
2596
+ // hard-to-reverse side effects a dry run promises never to do.
2597
+ if (options.issueSizeCheck) {
2598
+ const warnings = issueSizeWarnings(issues)
2599
+ const ownerRepo = options.autoSplit && !options.issuesJson && !options.dryRun ? originOwnerRepo(repo) : undefined
2600
+ if (warnings.length > 0 && ownerRepo) {
2601
+ const model = options.model ?? DEFAULT_MODEL
2602
+ const llmClient = new OpenRouterClient({ apiKey: process.env.HEADLESSCODE_OPENROUTER_API_KEY })
2603
+ const { issues: split, outcomes } = await autoSplitOversizedIssues(issues, warnings, {
2604
+ proposeSplit: (issue) => proposeSemanticSplit(issue, llmClient, model),
2605
+ createIssue: (issue) => createGhIssue(repo, ownerRepo, issue),
2606
+ closeParent: (n, comment) => closeGhIssue(repo, ownerRepo, n, comment),
2607
+ })
2608
+ issues = split
2609
+ for (const o of outcomes) {
2610
+ if (o.outcome === "split") {
2611
+ process.stdout.write(
2612
+ `[orchestrate] auto-split #${o.number} ("${o.title}") into ${o.created?.length} sub-issue(s): ` +
2613
+ (o.created ?? []).map((c) => `#${c.number} (${c.url})`).join(", ") +
2614
+ ` — parent closed\n`,
2615
+ )
2616
+ } else if (o.outcome === "kept-as-is") {
2617
+ process.stdout.write(
2618
+ `[orchestrate] #${o.number} ("${o.title}") flagged by the size heuristic, but the model determined ` +
2619
+ `it's one coherent piece of work — dispatching as-is\n`,
2620
+ )
2621
+ } else {
2622
+ process.stderr.write(
2623
+ `headlesscode orchestrate: WARNING: auto-split failed for #${o.number} ("${o.title}"): ${o.reason} — ` +
2624
+ `dispatching the original issue as-is. --no-auto-split silences future attempts.\n`,
2625
+ )
2626
+ }
2627
+ }
2628
+ } else if (warnings.length > 0) {
2629
+ printIssueSizeWarnings(warnings)
2630
+ }
2631
+ }
2632
+
2633
+ // Naming collision avoidance: a still-running round in the same repo
2634
+ // already occupies some `.worktrees/wN` dirs. Scan them and let
2635
+ // splitIssues name this round's groups around the gap instead of always
2636
+ // starting at w1 and colliding — see split.ts's occupiedNames param. Two
2637
+ // concurrent `orchestrate` invocations against the same repo can now
2638
+ // share it instead of the second one hard-failing on a "stale" worktree
2639
+ // that was actually just in-flight, not stale.
2640
+ let occupiedNames: Set<string> = new Set()
2641
+ try {
2642
+ occupiedNames = new Set(
2643
+ fs
2644
+ .readdirSync(path.join(repo, ".worktrees"), { withFileTypes: true })
2645
+ .filter((d) => d.isDirectory())
2646
+ .map((d) => d.name),
2647
+ )
2648
+ } catch {
2649
+ // No .worktrees dir yet — nothing occupied.
2650
+ }
2651
+ const specs = splitIssues(issues, { occupiedNames })
2652
+
2653
+ // Pre-spawn collision check: defense in depth only now that naming skips
2654
+ // occupied slots above — this should not fire in the normal case. It
2655
+ // still catches races (a worktree appearing between the scan above and
2656
+ // spawn) and fails loudly before spawning anything or writing task
2657
+ // files/state, so a skipped-at-spawn group can never be silently dropped.
2658
+ //
2659
+ // Liveness-aware (a real production incident, not hypothetical): a
2660
+ // generic "stale worktree" message previously read as an invitation to
2661
+ // manually `rm -rf`/`git worktree remove` the path — which, when the
2662
+ // worktree actually belonged to a still-running worker, destroyed live
2663
+ // session state mid-run and wasted real API spend, repeatedly, across
2664
+ // one dispatch session. Distinguish "still running" from "leftover" via
2665
+ // the SAME `.harness.pid` liveness check the watcher's own stall guard
2666
+ // uses (isPidAlive), and steer toward `orchestrate cleanup --apply`
2667
+ // specifically — NOT raw `git worktree remove`, which refuses on any
2668
+ // untracked file and can leave a stray `.headlesscode/` behind that
2669
+ // re-trips this exact check on the next attempt; `cleanup --apply`
2670
+ // removes harness artifacts (including `.headlesscode/`) and known-safe
2671
+ // residue BEFORE calling `git worktree remove`, precisely to avoid that
2672
+ // residue (see cleanup.ts's removeHarnessArtifacts/removeKnownSafeArtifacts).
2673
+ const staleWorktrees = specs
2674
+ .map((spec) => path.join(repo, ".worktrees", spec.name))
2675
+ .filter((wtPath) => fs.existsSync(wtPath))
2676
+ if (staleWorktrees.length > 0) {
2677
+ const details = staleWorktrees
2678
+ .map((wtPath) => {
2679
+ const rel = path.relative(repo, wtPath)
2680
+ return isPidAlive(wtPath) ? `${rel} (round IN PROGRESS — live pid)` : `${rel} (no live pid — likely leftover)`
2681
+ })
2682
+ .join(", ")
2683
+ process.stderr.write(
2684
+ `headlesscode orchestrate: worktree collision found: ${details}. ` +
2685
+ `If any are marked "round IN PROGRESS", do NOT delete them — that destroys a live session's state and spend. ` +
2686
+ `Retry once it finishes, or investigate with "headlesscode orchestrate status --repo ${repo}". ` +
2687
+ `For genuinely leftover worktrees, use "headlesscode orchestrate cleanup --repo ${repo} --apply" ` +
2688
+ `(only removes worktrees whose branch is verified merged, and clears .headlesscode/ before ` +
2689
+ `"git worktree remove" so it can't leave residue behind) — avoid raw "git worktree remove"/"rm -rf", ` +
2690
+ `which can leave a stray .headlesscode/ dir that re-trips this same check next time.\n`,
2691
+ )
2692
+ return 1
2693
+ }
2694
+
2695
+ if (options.dryRun) {
2696
+ // Per-mode model assignment: resolve each role's model exactly like a
2697
+ // real round so the dry-run preview shows the effective model per role
2698
+ // (mode-models.json / _default / OPENROUTER_MODEL; explicit --model
2699
+ // always wins). undefined -> the OpenRouter client's built-in default.
2700
+ const previewModel = (mode: string): string =>
2701
+ resolveModelForMode({ workspaceRoot: repo, mode, explicitModel: options.model, env: process.env }) ??
2702
+ "deepseek/deepseek-v4-flash-0731 (client default)"
2703
+ // Issue #16: estimate the round's expected cost/iterations from
2704
+ // recorded cost history, keyed by issue shape. Advisory only — a
2705
+ // history read failure must never fail the dry-run, just omit the
2706
+ // section.
2707
+ let historyRecords: Awaited<ReturnType<typeof readCostHistory>> = []
2708
+ try {
2709
+ historyRecords = await readCostHistory(repo)
2710
+ } catch {
2711
+ historyRecords = []
2712
+ }
2713
+ const estimateSection = buildEstimateSection(estimateGroups(historyRecords, specs, issues), historyRecords.length)
2714
+ printPlan(repo, options.mode, specs, issues, {
2715
+ worker: previewModel(options.mode),
2716
+ reviewer: previewModel(options.reviewMode),
2717
+ qa: previewModel(options.qaMode),
2718
+ }, options.maxIterations, estimateSection,
2719
+ options.planFirst
2720
+ ? { mode: options.planFirstMode, maxIterations: options.planFirstMaxIterations }
2721
+ : undefined,
2722
+ )
2723
+ return 0
2724
+ }
2725
+
2726
+ if (!process.env.HEADLESSCODE_OPENROUTER_API_KEY) {
2727
+ process.stderr.write(
2728
+ "headlesscode orchestrate: HEADLESSCODE_OPENROUTER_API_KEY is not set (workers + reviewer need it).\n" +
2729
+ " Use --dry-run to preview the plan without an API key.\n",
2730
+ )
2731
+ return 2
2732
+ }
2733
+
2734
+ // 1a. Phase 6 GLOBAL concurrency cap (spec 6.3): abort BEFORE any work
2735
+ // (task files, spawn) when the fleet is already at/over the cap.
2736
+ // Cross-process view: orchestrator groups in spawned/running + watcher
2737
+ // in-flight 'spawned' entries. Fail-fast by design — orchestrate never
2738
+ // queues silently (the watcher's durable 'pending' state is the queueing
2739
+ // mechanism; see docs/phase6-cloud.md).
2740
+ const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
2741
+ const activeNow = activeSessionCountForRepo(repo)
2742
+ if (activeNow >= options.maxConcurrentSessions) {
2743
+ process.stderr.write(
2744
+ `headlesscode orchestrate: concurrency cap reached — ${activeNow} active session(s) ` +
2745
+ `(cap ${options.maxConcurrentSessions}, set via --max-concurrent-sessions or ` +
2746
+ `HEADLESSCODE_MAX_CONCURRENT_SESSIONS). This round was NOT spawned. ` +
2747
+ `Wait for running sessions to finish or raise the cap.\n`,
2748
+ )
2749
+ return 1
2750
+ }
2751
+
2752
+ // 1a-sync. Issue #25: local master silently drifts unpushed from origin,
2753
+ // and every PR-merge cycle pays for it with re-merge conflict cascades.
2754
+ // At this "a new round is about to start" checkpoint, reconcile the two
2755
+ // (fetch origin, push local-only commits, fast-forward local onto origin)
2756
+ // so this round's branches start from what's actually on GitHub. Best-
2757
+ // effort auxiliary step: a failure is a loud warning, never an abort.
2758
+ // HEADLESSCODE_ORCHESTRATE_NO_SYNC=1 disables the mutation and only warns
2759
+ // on non-trivial drift (see git-sync.ts).
2760
+ if (process.env[ORCHESTRATE_SYNC_DISABLED_ENV]) {
2761
+ const drift = branchSyncStatus(repo)
2762
+ if (drift && drift.ahead > TRIVIAL_DRIFT_AHEAD) {
2763
+ process.stderr.write(
2764
+ `headlesscode orchestrate: WARNING: local ${drift.branch} is ${drift.ahead} commit(s) ahead of ${drift.remoteRef} ` +
2765
+ `(${ORCHESTRATE_SYNC_DISABLED_ENV} set — not pushing). Push it now to avoid merge-conflict cascades at ` +
2766
+ `PR-merge time (issue #25): git -C ${repo} push origin ${drift.branch}\n`,
2767
+ )
2768
+ }
2769
+ } else {
2770
+ const sync = syncBranchWithOrigin(repo)
2771
+ for (const line of syncSummaryLines(sync)) {
2772
+ process.stdout.write(`[sync] ${line}\n`)
2773
+ }
2774
+ for (const line of syncWarningLines(sync, repo)) {
2775
+ process.stderr.write(`headlesscode orchestrate: ${line}\n`)
2776
+ }
2777
+ }
2778
+
2779
+ // 1a-pre. Issue #13 pre-spawn preflight probe: a cheap 1-token completion
2780
+ // using the EXACT model + provider pin a real worker will use, so a dead
2781
+ // key / exhausted account balance / down pinned provider fails NOW instead
2782
+ // of 30-80 iterations (10-15 min + real spend) into the round. Runs before
2783
+ // task files or spawn; --no-preflight skips it for CI/non-interactive
2784
+ // contexts that don't want the extra round-trip. The probe's model is
2785
+ // resolved exactly like the worker's first spawn (same resolveModelForMode
2786
+ // precedence as 1c below), so what we probe is what the worker will call.
2787
+ let preflightProbe: PreflightResult | LocalPreflightResult | undefined
2788
+ if (options.preflight) {
2789
+ // 2026-08-27: probe whichever backend the WORKER's mode will actually
2790
+ // use (see cli.ts's useLocalCodeBackend / reviewer.ts's runReview /
2791
+ // qa.ts's runQa — same gate, repeated here since orchestrate never
2792
+ // otherwise imports cli.ts). Before this, a round configured entirely
2793
+ // for the local daemon still probed OpenRouter/DeepSeek unconditionally
2794
+ // and aborted on a cloud key that round would never touch.
2795
+ const useLocalBackendForWorker =
2796
+ process.env.HEADLESSCODE_CODE_MODE_BACKEND === "ollama" &&
2797
+ (process.env.HEADLESSCODE_LOCAL_BACKEND_MODES ?? "code")
2798
+ .split(",")
2799
+ .map((s) => s.trim())
2800
+ .filter(Boolean)
2801
+ .includes(options.mode)
2802
+ try {
2803
+ if (useLocalBackendForWorker) {
2804
+ preflightProbe = await runLocalPreflight({
2805
+ model: options.model ?? resolvePerModeEnv("HEADLESSCODE_CODE_MODE_MODEL", options.mode),
2806
+ baseUrl: resolvePerModeEnv("HEADLESSCODE_OLLAMA_URL", options.mode),
2807
+ })
2808
+ } else {
2809
+ const workerModelForProbe = resolveModelForMode({
2810
+ workspaceRoot: repo,
2811
+ mode: options.mode,
2812
+ explicitModel: options.model,
2813
+ env: process.env,
2814
+ })
2815
+ const maxCostRaw = process.env.HEADLESSCODE_MAX_COST_USD
2816
+ const maxCostUsd = maxCostRaw !== undefined && maxCostRaw !== "" ? Number(maxCostRaw) : undefined
2817
+ preflightProbe = await runPreflight({
2818
+ model: workerModelForProbe,
2819
+ workerSessions: specs.length,
2820
+ maxCostUsd: maxCostUsd !== undefined && Number.isFinite(maxCostUsd) && maxCostUsd > 0 ? maxCostUsd : undefined,
2821
+ })
2822
+ }
2823
+ } catch (err) {
2824
+ // Neither probe throws by contract, but if either ever does the
2825
+ // round must not proceed on an unchecked gate.
2826
+ process.stderr.write(
2827
+ `headlesscode orchestrate: preflight probe crashed unexpectedly: ${err instanceof Error ? err.message : String(err)}\n`,
2828
+ )
2829
+ return 1
2830
+ }
2831
+ process.stdout.write(`[preflight] ${preflightProbe.line}\n`)
2832
+ if (preflightProbe.status !== "ok") {
2833
+ process.stderr.write(
2834
+ `headlesscode orchestrate: preflight FAILED (${preflightProbe.status}) — aborting before any workers spawn. ` +
2835
+ `Fix the problem above, or use --no-preflight to skip this check.\n`,
2836
+ )
2837
+ return 1
2838
+ }
2839
+ }
2840
+
2841
+ // 1b. Task files.
2842
+ const written = writeTaskFiles(repo, specs, issues, { planFirst: options.planFirst })
2843
+ process.stdout.write(`Wrote ${written.length} task file(s) under ${path.join(repo, "plans", "parallel-tasks")}\n`)
2844
+
2845
+ // 1c. Per-mode model assignment for the WORKER'S FIRST SPAWN. This must
2846
+ // resolve BEFORE step 2's spawnSync: the spawner script (and through it
2847
+ // run-worker.sh) inherits OPENROUTER_MODEL from the environment, so a
2848
+ // mode-models.json entry for the worker's mode would otherwise be silently
2849
+ // ignored on the first attempt (it only kicked in on rework re-spawns).
2850
+ // resolveModelForMode's own precedence holds here: an explicit --model
2851
+ // flag beats the config file, which beats OPENROUTER_MODEL.
2852
+ const { env: spawnEnv, workerModel } = buildSpawnEnv({
2853
+ repo,
2854
+ mode: options.mode,
2855
+ explicitModel: options.model,
2856
+ memoryDir: options.memoryDir,
2857
+ maxIterations: options.maxIterations,
2858
+ planFirst: options.planFirst,
2859
+ planFirstMode: options.planFirstMode,
2860
+ planFirstMaxIterations: options.planFirstMaxIterations,
2861
+ env: process.env,
2862
+ })
2863
+
2864
+ // 2. Spawn (the bash script is the actual spawner; cwd = target repo so
2865
+ // `git rev-parse --show-toplevel` resolves there).
2866
+ const spawnCmd = `bash ${path.join(HARNESS_ROOT_TS, "scripts", "spawn-parallel-worktrees.sh")} ${specs
2867
+ .map((spec) => `${spec.name}:${specs.indexOf(spec)}:plans/parallel-tasks/${spec.taskFile}`)
2868
+ .join(" ")}`
2869
+ process.stdout.write(`Spawn: ${spawnCmd}\n`)
2870
+ // `bash -c <spawnCmd>` (NOT spawnSync("bash", [spawnCmd], { shell: true }),
2871
+ // which would run `bash bash <script> …` and die with "cannot execute
2872
+ // binary file").
2873
+ const spawnResult = spawnSync("bash", ["-c", spawnCmd], {
2874
+ cwd: repo,
2875
+ env: spawnEnv,
2876
+ stdio: "inherit",
2877
+ })
2878
+ if (spawnResult.status !== 0) {
2879
+ process.stderr.write(`headlesscode orchestrate: spawn script failed (exit ${spawnResult.status})\n`)
2880
+ return 1
2881
+ }
2882
+
2883
+ // 3. Record batch + groups in the state file (the spawner already wrote
2884
+ // group entries; add the batch id).
2885
+ const batch = options.batch ?? `round-${new Date().toISOString().slice(0, 10)}`
2886
+ try {
2887
+ // Issue #80: the whole load-merge-save below runs inside mutateState's
2888
+ // cross-process lock. Without it, two concurrent `orchestrate` invocations
2889
+ // against the same repo (explicitly supported — see the concurrency
2890
+ // budget check above) can both loadState the same on-disk snapshot, then
2891
+ // whichever saveState runs last silently discards the other's round's
2892
+ // groups. patchGroup already guards single-group patches this way; this
2893
+ // is the same guarantee for this multi-group initial-write transaction.
2894
+ await mutateState(statePath, async (fresh) => {
2895
+ let state = fresh
2896
+ state.batch = batch
2897
+ // Issue #16: stamp every group with its per-issue shape (split.ts's
2898
+ // issueShape) at dispatch time — the only moment the issue titles/bodies
2899
+ // are in hand. cost-history recording happens LATER (when the group
2900
+ // reaches a terminal status, possibly in a separate orchestrate
2901
+ // invocation) and only has issue numbers; the shape is what lets
2902
+ // cost-estimate.ts match a NEW task to similar past ones, so it must be
2903
+ // captured here and persisted in the state file (OrchestratorGroup.shapes).
2904
+ // The titles/bodies themselves are captured at the same moment into
2905
+ // OrchestratorGroup.issueBodies: the rework/QA/continuation task files are
2906
+ // generated LATER from state alone (a separate orchestrate invocation) and
2907
+ // need them to embed each issue's real title/body inline instead of a
2908
+ // runtime `gh issue view` (which fails for synthetic --issues-json
2909
+ // numbers).
2910
+ const shapeByNumber = new Map<number, string>()
2911
+ const issueBodiesByNumber = new Map<number, { title: string; body?: string }>()
2912
+ for (const issue of issues) {
2913
+ shapeByNumber.set(issue.number, issueShape(issue))
2914
+ issueBodiesByNumber.set(issue.number, { title: issue.title, body: issue.body })
2915
+ }
2916
+ const groupNamesOnDisk = new Set(state.groups.map((g) => g.name))
2917
+ for (const spec of specs) {
2918
+ if (!groupNamesOnDisk.has(spec.name)) {
2919
+ continue
2920
+ }
2921
+ const shapes = spec.issues.map((n) => shapeByNumber.get(n) ?? "generic")
2922
+ const issueBodies: IssueBodies = {}
2923
+ for (const n of spec.issues) {
2924
+ const entry = issueBodiesByNumber.get(n)
2925
+ if (entry) {
2926
+ issueBodies[String(n)] = entry
2927
+ }
2928
+ }
2929
+ state = updateGroup(state, spec.name, { shapes, issueBodies })
2930
+ }
2931
+ // Issue #13: persist the preflight probe so the dashboard can surface the
2932
+ // preflight line on the round view (see aggregate.ts RoundSummary.preflight).
2933
+ if (preflightProbe) {
2934
+ // LocalPreflightResult has no cost fields at all (local inference
2935
+ // is genuinely free) — default to 0/undefined rather than widen
2936
+ // PreflightRecord's required probeCostUsd to optional for a case
2937
+ // that's always a real number either way.
2938
+ state.preflight = {
2939
+ status: preflightProbe.status,
2940
+ model: preflightProbe.model,
2941
+ baseUrl: preflightProbe.baseUrl,
2942
+ line: preflightProbe.line,
2943
+ latencyMs: preflightProbe.latencyMs,
2944
+ probeCostUsd: "probeCostUsd" in preflightProbe ? preflightProbe.probeCostUsd : 0,
2945
+ roundCostEstimateUsd: "roundCostEstimateUsd" in preflightProbe ? preflightProbe.roundCostEstimateUsd : undefined,
2946
+ }
2947
+ }
2948
+ return state
2949
+ })
2950
+ } catch (err) {
2951
+ if (!(err instanceof SyntaxError)) {
2952
+ // Not a corrupt-JSON case (e.g. the cross-process lock timed out) —
2953
+ // nothing to back up, just report and abort loudly.
2954
+ process.stderr.write(
2955
+ `headlesscode orchestrate: failed to record round state in ${statePath}: ` +
2956
+ `${err instanceof Error ? err.message : String(err)}\n`,
2957
+ )
2958
+ return 1
2959
+ }
2960
+ // Issue #77: loadState (inside mutateState) only throws SyntaxError for a
2961
+ // genuinely corrupt/truncated state file (ENOENT is already handled
2962
+ // inside loadState and returns a fresh default). Silently replacing it
2963
+ // here would destroy every group's status/review verdict with no
2964
+ // warning. Back the corrupt file up so it isn't lost, then refuse to
2965
+ // start rather than clobber it.
2966
+ const backupPath = `${statePath}.corrupt-${Date.now()}`
2967
+ try {
2968
+ await fs.promises.rename(statePath, backupPath)
2969
+ } catch {
2970
+ /* best-effort backup; fall through to the loud abort either way */
2971
+ }
2972
+ process.stderr.write(
2973
+ `headlesscode orchestrate: state file ${statePath} is corrupt and could not be parsed ` +
2974
+ `(${err.message}). Backed up to ${backupPath}. ` +
2975
+ `Refusing to start with an empty state — restore or repair the backup, then retry.\n`,
2976
+ )
2977
+ return 1
2978
+ }
2979
+
2980
+ // 4. Watch + review. Per-mode model assignment: each role resolves its OWN
2981
+ // model — workers use the group's mode (resolved in 1c, above — the SAME
2982
+ // const reused here for the rework loop), the reviewer uses reviewMode, QA
2983
+ // uses qaMode. An explicit --model flag beats every mode's config entry
2984
+ // (resolveModelForMode precedence rule 1), so a blanket override still
2985
+ // works exactly as before.
2986
+ const reviewerModel = resolveModelForMode({
2987
+ workspaceRoot: repo,
2988
+ mode: options.reviewMode,
2989
+ explicitModel: options.model,
2990
+ env: process.env,
2991
+ })
2992
+ const qaModel = resolveModelForMode({
2993
+ workspaceRoot: repo,
2994
+ mode: options.qaMode,
2995
+ explicitModel: options.model,
2996
+ env: process.env,
2997
+ })
2998
+
2999
+ process.stdout.write("Watching for worker completion (.harness.done markers)...\n")
3000
+ await watchGroups({
3001
+ repoRoot: repo,
3002
+ statePath,
3003
+ pollIntervalMs: options.pollIntervalMs,
3004
+ reviewEnabled: options.review,
3005
+ qaEnabled: options.qa,
3006
+ onGroupUpdate: async (group) => {
3007
+ // Auto-continue on iteration-exhaustion (issue #2 Part 2): a worker
3008
+ // that failed ONLY because it hit --max-iterations gets a fresh
3009
+ // session on the SAME worktree (the partial state is on disk), up
3010
+ // to --max-continuations, then the group is marked needs-human.
3011
+ // Runs regardless of --no-review (this is about worker failure, not
3012
+ // review routing) and never triggers for budget stops or real
3013
+ // errors — handleIterationExhaustion's empty patch keeps "failed".
3014
+ if (group.status === "failed") {
3015
+ const decision = handleIterationExhaustion(
3016
+ group,
3017
+ repo,
3018
+ options.maxContinuations,
3019
+ options.mode,
3020
+ workerModel,
3021
+ options.maxIterations,
3022
+ )
3023
+ if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
3024
+ fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
3025
+ fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
3026
+ process.stdout.write(
3027
+ `[orchestrate] ${group.name} continuation ${decision.newContinuationCount}: ` +
3028
+ `previous worker hit the iteration cap; wrote ${path.relative(repo, decision.taskFilePath)}; ` +
3029
+ `re-spawning worker on the same worktree...\n`,
3030
+ )
3031
+ const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
3032
+ cwd: repo,
3033
+ env: {
3034
+ ...process.env,
3035
+ HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3036
+ ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3037
+ },
3038
+ stdio: "inherit",
3039
+ })
3040
+ if (spawnResult.status !== 0) {
3041
+ // The continuation worker could not be launched — the
3042
+ // group cannot continue itself, so surface it as needing
3043
+ // a human instead of leaving a dangling "running" group.
3044
+ process.stderr.write(
3045
+ `[orchestrate] continuation spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
3046
+ `marking group needs-human\n`,
3047
+ )
3048
+ // patchGroup reloads the state file fresh immediately
3049
+ // before merging (under a write lock) rather than basing
3050
+ // the merge on `currentState` — a stale snapshot that may
3051
+ // predate other groups' concurrent writes now that
3052
+ // review/QA for multiple groups can run in parallel
3053
+ // (issue #24).
3054
+ await patchGroup(statePath, group.name, {
3055
+ status: "needs-human",
3056
+ continuationCount: decision.newContinuationCount,
3057
+ last_activity: {
3058
+ note: `continuation spawn failed after ${decision.newContinuationCount} attempt(s); needs human`,
3059
+ },
3060
+ })
3061
+ } else {
3062
+ // Spawn succeeded: reset the group so the watcher re-polls
3063
+ // it as "running" → "done/failed" → re-continued exactly
3064
+ // like a first attempt.
3065
+ await patchGroup(statePath, group.name, decision.patch)
3066
+ }
3067
+ return true
3068
+ }
3069
+ if (decision.patch.status === "needs-human") {
3070
+ // Continuation cap reached: terminal "needs-human" outcome.
3071
+ await patchGroup(statePath, group.name, decision.patch)
3072
+ process.stderr.write(
3073
+ `[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newContinuationCount} continuation ` +
3074
+ `attempt(s) and still hits the iteration cap — a human must look at this worktree.\n`,
3075
+ )
3076
+ return true
3077
+ }
3078
+ // Real failure (not iteration exhaustion): stays failed.
3079
+ return
3080
+ }
3081
+
3082
+ // Deterministic post-hoc log analysis: runs once per group as soon
3083
+ // as it reaches "done", unconditionally (no --review/--qa gate, no
3084
+ // LLM cost) — the same tool-call/error/stall/repeated-command
3085
+ // findings a human would otherwise get by hand-reading harness.log
3086
+ // + events.jsonl. Never blocks review/QA below; a failure here is
3087
+ // logged and swallowed.
3088
+ if (group.status === "done" && group.log_analysis === undefined) {
3089
+ try {
3090
+ const analysis = await analyzeWorktreeSessions(groupWorktreePath(repo, group))
3091
+ const patch = {
3092
+ log_analysis: analysis
3093
+ ? {
3094
+ findings: analysis.findings,
3095
+ toolCallCounts: analysis.toolCallCounts,
3096
+ toolErrorCounts: analysis.toolErrorCounts,
3097
+ analyzedAt: new Date().toISOString(),
3098
+ }
3099
+ : (false as const),
3100
+ }
3101
+ await patchGroup(statePath, group.name, patch)
3102
+ if (analysis && analysis.findings.length > 0) {
3103
+ process.stdout.write(
3104
+ `[orchestrate] ${group.name} log analysis: ${analysis.findings.length} finding(s):\n` +
3105
+ analysis.findings.map((f) => ` - ${f}\n`).join(""),
3106
+ )
3107
+ }
3108
+ } catch (err) {
3109
+ process.stderr.write(
3110
+ `[orchestrate] log analysis of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
3111
+ )
3112
+ }
3113
+ }
3114
+
3115
+ if (!options.review) {
3116
+ return
3117
+ }
3118
+ if (group.status !== "done" || group.review_verdict !== undefined) {
3119
+ return
3120
+ }
3121
+ process.stdout.write(`[orchestrate] reviewing ${group.name} (branch ${group.branch ?? "?"})...\n`)
3122
+ try {
3123
+ const reviewWtPath = groupWorktreePath(repo, group)
3124
+ // See hasRealWorktreeChanges's doc comment (Tier 1 of the
3125
+ // 2026-08-28 fabricated-review incident fix): a "clean" verdict
3126
+ // is structurally impossible with zero real changes, checked
3127
+ // directly against git — never asked of (or trusted from) the
3128
+ // review LLM itself. Skips the review session entirely rather
3129
+ // than spend a call that has nothing real to verify.
3130
+ if (!hasRealWorktreeChanges(reviewWtPath)) {
3131
+ await patchGroup(statePath, group.name, {
3132
+ review_verdict: "finding",
3133
+ pending_review_findings: [
3134
+ "No real changes found in the worktree (empty diff vs the tracked upstream, no uncommitted changes outside harness bookkeeping files) — there is nothing for a review to verify. Automatic fail: a \"clean\" verdict is structurally impossible with zero changes.",
3135
+ ],
3136
+ reviewed_at: new Date().toISOString(),
3137
+ last_activity: { note: "review skipped: worktree has no real changes (automatic fail)" },
3138
+ })
3139
+ process.stdout.write(
3140
+ `[orchestrate] AUTOMATIC FAIL: ${group.name}'s worktree has no real changes — skipping the review session (a "clean" verdict is structurally impossible with an empty diff)\n`,
3141
+ )
3142
+ return true
3143
+ }
3144
+ const rawResult = await runReviewWithRetries({
3145
+ workspaceRoot: reviewWtPath,
3146
+ mode: options.reviewMode,
3147
+ model: reviewerModel,
3148
+ issues: group.issues,
3149
+ })
3150
+ // A review-SESSION failure must not feed into the rework-a-worker
3151
+ // path below — see handleReviewSessionError's doc comment.
3152
+ if (rawResult.verdict === "error") {
3153
+ await patchGroup(statePath, group.name, handleReviewSessionError(rawResult))
3154
+ process.stderr.write(
3155
+ `[orchestrate] NEEDS-HUMAN: ${group.name}'s review session failed repeatedly — ` +
3156
+ `${rawResult.summary}\n`,
3157
+ )
3158
+ return true
3159
+ }
3160
+ // See hasRealVerificationActivity's doc comment (Tier 2 of the
3161
+ // 2026-08-28 incident fix): a "clean" claim backed by zero real,
3162
+ // successful execute_command results in the session's OWN
3163
+ // transcript is downgraded to a finding rather than trusted —
3164
+ // the exact gap that let a fabricated review through even with
3165
+ // the required structured verdict line.
3166
+ let result = rawResult
3167
+ if (rawResult.verdict === "clean" && !hasRealVerificationActivity(reviewWtPath, rawResult.reportPath)) {
3168
+ process.stdout.write(
3169
+ `[orchestrate] note: ${group.name}'s review verdict was "clean" but the session's own transcript shows no real, successful execute_command result — downgrading to a finding rather than trust an unverified claim\n`,
3170
+ )
3171
+ result = {
3172
+ ...rawResult,
3173
+ verdict: "finding",
3174
+ findings: [
3175
+ ...rawResult.findings,
3176
+ "Review declared \"clean\" but its own transcript shows no real, successful execute_command result — nothing to substantiate the verdict was actually run. Treated as a finding rather than trusted.",
3177
+ ],
3178
+ }
3179
+ }
3180
+ const patch = {
3181
+ review_verdict: result.verdict,
3182
+ pending_review_findings: result.findings,
3183
+ reviewed_at: new Date().toISOString(),
3184
+ // Issue #34: point at the review session's complete final
3185
+ // report so the full reasoning behind the verdict is one
3186
+ // file-read away, not a re-run away.
3187
+ review_report: result.reportPath,
3188
+ last_activity: {
3189
+ note: `reviewed: verdict=${result.verdict} (${result.findings.length} finding(s))`,
3190
+ },
3191
+ }
3192
+ await patchGroup(statePath, group.name, patch)
3193
+ process.stdout.write(
3194
+ `[orchestrate] ${group.name} review verdict: ${result.verdict} (${result.findings.length} finding(s))\n`,
3195
+ )
3196
+
3197
+ // Rework loop (plans/rework-loop.md): a "finding" verdict is
3198
+ // treated as NEW WORK — re-spawn a worker on the SAME worktree to
3199
+ // fix the findings, up to --max-rework-cycles attempts. Returning
3200
+ // true tells watchGroups to reload state from disk so it re-polls
3201
+ // the group as "running" → "done" → re-reviewed (the watcher's
3202
+ // in-memory state is otherwise stale — see watchGroups).
3203
+ if (result.verdict === "finding") {
3204
+ const decision = handleReviewVerdict(
3205
+ group,
3206
+ result,
3207
+ repo,
3208
+ options.maxReworkCycles,
3209
+ options.mode,
3210
+ workerModel,
3211
+ options.maxIterations,
3212
+ )
3213
+ if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
3214
+ fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
3215
+ fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
3216
+ process.stdout.write(
3217
+ `[orchestrate] ${group.name} rework cycle ${decision.newReworkCount}: ` +
3218
+ `wrote ${path.relative(repo, decision.taskFilePath)}; re-spawning worker on the same worktree...\n`,
3219
+ )
3220
+ const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
3221
+ cwd: repo,
3222
+ env: {
3223
+ ...process.env,
3224
+ HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3225
+ ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3226
+ },
3227
+ stdio: "inherit",
3228
+ })
3229
+ if (spawnResult.status !== 0) {
3230
+ // The rework worker could not be launched — the group
3231
+ // cannot fix itself, so surface it as needing a human
3232
+ // instead of leaving a dangling "running" group.
3233
+ process.stderr.write(
3234
+ `[orchestrate] rework worker spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
3235
+ `marking group needs-human\n`,
3236
+ )
3237
+ await patchGroup(statePath, group.name, {
3238
+ status: "needs-human",
3239
+ reworkCount: decision.newReworkCount,
3240
+ last_activity: {
3241
+ note: `rework spawn failed after ${decision.newReworkCount} attempt(s); needs human`,
3242
+ },
3243
+ })
3244
+ } else {
3245
+ // Spawn succeeded: reset the group so the watcher
3246
+ // re-polls it as "running" → "done" → re-reviewed.
3247
+ await patchGroup(statePath, group.name, decision.patch)
3248
+ }
3249
+ return true
3250
+ }
3251
+ // Cap reached: terminal "needs-human" outcome. Findings stay
3252
+ // recorded so a human can see exactly what the reviewer flagged.
3253
+ await patchGroup(statePath, group.name, decision.patch)
3254
+ process.stderr.write(
3255
+ `[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newReworkCount} rework ` +
3256
+ `attempt(s) and the review still has findings — a human must look at this worktree.\n`,
3257
+ )
3258
+ return true
3259
+ }
3260
+ } catch (err) {
3261
+ process.stderr.write(
3262
+ `[orchestrate] review of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
3263
+ )
3264
+ }
3265
+ // Phase 4: after a group's workers complete AND review passes, run a
3266
+ // headless QA session against that worktree. QA only runs when review
3267
+ // was clean (or review was disabled); the result is recorded in the
3268
+ // group's `qa` field. QA uses read+command tools only — it can boot
3269
+ // the app and run tests but cannot modify source files.
3270
+ if (!options.qa) {
3271
+ return
3272
+ }
3273
+ if (group.status !== "done" || group.qa !== undefined) {
3274
+ return
3275
+ }
3276
+ // The callback's `group` is the pre-review snapshot; the review
3277
+ // verdict lives in the state file we just persisted via patchGroup,
3278
+ // so read it fresh from disk rather than any in-memory copy —
3279
+ // review/QA for other groups can be writing concurrently now
3280
+ // (issue #24), so a captured snapshot here could be stale.
3281
+ const updatedGroup = loadStateSync(statePath).groups.find((g) => g.name === group.name)
3282
+ const reviewPassed = !options.review || updatedGroup?.review_verdict === "clean"
3283
+ if (!reviewPassed) {
3284
+ return
3285
+ }
3286
+ process.stdout.write(`[orchestrate] QA ${group.name} (branch ${group.branch ?? "?"})...\n`)
3287
+ try {
3288
+ const qaWtPath = groupWorktreePath(repo, group)
3289
+ // See hasRealWorktreeChanges's doc comment: same automatic-fail
3290
+ // gate as the review step above, checked directly against git
3291
+ // before the QA LLM session ever runs — a "pass" verdict is
3292
+ // structurally impossible with zero real changes.
3293
+ if (!hasRealWorktreeChanges(qaWtPath)) {
3294
+ await patchGroup(statePath, group.name, {
3295
+ qa: {
3296
+ status: "failed",
3297
+ verdict: "fail",
3298
+ evidence:
3299
+ "No real changes found in the worktree (empty diff vs the tracked upstream, no uncommitted changes outside harness bookkeeping files) — there is nothing for QA to verify. Automatic fail: a \"pass\" verdict is structurally impossible with zero changes.",
3300
+ updated: new Date().toISOString(),
3301
+ },
3302
+ last_activity: { note: "QA skipped: worktree has no real changes (automatic fail)" },
3303
+ })
3304
+ process.stdout.write(
3305
+ `[orchestrate] AUTOMATIC FAIL: ${group.name}'s worktree has no real changes — skipping the QA session (a "pass" verdict is structurally impossible with an empty diff)\n`,
3306
+ )
3307
+ return true
3308
+ }
3309
+ const rawQaResult = await runQaWithRetries({
3310
+ workspaceRoot: qaWtPath,
3311
+ mode: options.qaMode,
3312
+ model: qaModel,
3313
+ })
3314
+ // A QA-SESSION failure (crash/budget/mistake-limit/no result —
3315
+ // see runQaWithRetries) survived every retry: verdict "error",
3316
+ // distinct from a real "fail". Must escalate to needs-human, NOT
3317
+ // silently record a "failed"-looking QA with empty evidence and
3318
+ // leave the group's top-level status "done" — that would settle
3319
+ // the round (cost recorded, nothing left to retry) while QA never
3320
+ // actually validated anything. Caught live 2026-08-05 on issue
3321
+ // #18's own round: the QA-side twin of #33's review bug.
3322
+ if (rawQaResult.verdict === "error") {
3323
+ await patchGroup(statePath, group.name, handleQaSessionError(rawQaResult))
3324
+ process.stderr.write(
3325
+ `[orchestrate] NEEDS-HUMAN: ${group.name}'s QA session failed repeatedly — ` +
3326
+ `${rawQaResult.summary}\n`,
3327
+ )
3328
+ return
3329
+ }
3330
+ // See hasRealVerificationActivity's doc comment (Tier 2 of the
3331
+ // 2026-08-28 incident fix): same downgrade as the review step
3332
+ // above — a "pass" claim backed by zero real, successful
3333
+ // execute_command results in the session's OWN transcript is
3334
+ // downgraded to fail (and routed through the same rework loop
3335
+ // as a real QA fail below) rather than trusted.
3336
+ let qaResult = rawQaResult
3337
+ if (rawQaResult.verdict === "pass" && !hasRealVerificationActivity(qaWtPath, rawQaResult.reportPath)) {
3338
+ process.stdout.write(
3339
+ `[orchestrate] note: ${group.name}'s QA verdict was "pass" but the session's own transcript shows no real, successful execute_command result — downgrading to fail rather than trust an unverified claim\n`,
3340
+ )
3341
+ qaResult = {
3342
+ ...rawQaResult,
3343
+ verdict: "fail",
3344
+ evidence:
3345
+ `QA declared "pass" but its own transcript shows no real, successful execute_command result — ` +
3346
+ `nothing to substantiate the verdict was actually run. Treated as fail rather than trusted.\n\n` +
3347
+ `Original evidence: ${rawQaResult.evidence}`,
3348
+ }
3349
+ }
3350
+ // A real QA FAIL verdict is NEW WORK, not a settled "done"
3351
+ // (issue #52, caught live 2026-08-05): the group passed review
3352
+ // but QA caught real problems. Auto-spawn a rework cycle on the
3353
+ // SAME worktree exactly like a review finding, feeding the QA
3354
+ // evidence into the task file instead of review findings, up to
3355
+ // --max-rework-cycles. At the cap the group goes terminal
3356
+ // needs-human — it must never silently settle as "done" with
3357
+ // only the nested qa.status showing "failed".
3358
+ if (qaResult.verdict === "fail") {
3359
+ const decision = handleQaVerdict(
3360
+ group,
3361
+ qaResult,
3362
+ repo,
3363
+ options.maxReworkCycles,
3364
+ options.mode,
3365
+ workerModel,
3366
+ options.maxIterations,
3367
+ )
3368
+ if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
3369
+ fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
3370
+ fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
3371
+ process.stdout.write(
3372
+ `[orchestrate] ${group.name} QA-fail rework cycle ${decision.newReworkCount}: ` +
3373
+ `wrote ${path.relative(repo, decision.taskFilePath)}; re-spawning worker on the same worktree...\n`,
3374
+ )
3375
+ const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
3376
+ cwd: repo,
3377
+ env: {
3378
+ ...process.env,
3379
+ HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
3380
+ ...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
3381
+ },
3382
+ stdio: "inherit",
3383
+ })
3384
+ if (spawnResult.status !== 0) {
3385
+ // The QA-fail rework worker could not be launched —
3386
+ // the group cannot fix itself, so surface it as
3387
+ // needing a human instead of leaving a dangling
3388
+ // "running" group (same as the review-finding path).
3389
+ process.stderr.write(
3390
+ `[orchestrate] QA-fail rework worker spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
3391
+ `marking group needs-human\n`,
3392
+ )
3393
+ await patchGroup(statePath, group.name, {
3394
+ status: "needs-human",
3395
+ reworkCount: decision.newReworkCount,
3396
+ qa: {
3397
+ status: "failed",
3398
+ verdict: qaResult.verdict,
3399
+ evidence: qaResult.evidence.slice(0, 4000),
3400
+ report: qaResult.reportPath,
3401
+ updated: new Date().toISOString(),
3402
+ },
3403
+ last_activity: {
3404
+ note: `QA-fail rework spawn failed after ${decision.newReworkCount} attempt(s); needs human`,
3405
+ },
3406
+ })
3407
+ } else {
3408
+ // Spawn succeeded: reset the group so the watcher
3409
+ // re-polls it as "running" → "done" → re-reviewed
3410
+ // → re-QA'd.
3411
+ await patchGroup(statePath, group.name, decision.patch)
3412
+ }
3413
+ return true
3414
+ }
3415
+ // Cap reached: terminal "needs-human" outcome. The QA
3416
+ // verdict/evidence stay recorded so a human can see
3417
+ // exactly what failed.
3418
+ await patchGroup(statePath, group.name, decision.patch)
3419
+ process.stderr.write(
3420
+ `[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newReworkCount} rework ` +
3421
+ `attempt(s) and QA still fails — a human must look at this worktree.\n`,
3422
+ )
3423
+ return true
3424
+ }
3425
+ await patchGroup(statePath, group.name, {
3426
+ qa: {
3427
+ status: "done",
3428
+ verdict: "pass",
3429
+ evidence: qaResult.evidence.slice(0, 4000),
3430
+ // Issue #34: the state file keeps the lightweight parsed
3431
+ // fields for quick scanning; point at the QA session's
3432
+ // COMPLETE final report so the full reasoning behind the
3433
+ // verdict is one file-read away, not a re-run away.
3434
+ report: qaResult.reportPath,
3435
+ updated: new Date().toISOString(),
3436
+ },
3437
+ last_activity: {
3438
+ note: "QA: verdict=pass",
3439
+ },
3440
+ })
3441
+ process.stdout.write(
3442
+ `[orchestrate] ${group.name} QA verdict: pass (status done)\n`,
3443
+ )
3444
+ } catch (err) {
3445
+ process.stderr.write(
3446
+ `[orchestrate] QA of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
3447
+ )
3448
+ }
3449
+ },
3450
+ })
3451
+
3452
+ // Reload the state from disk: onGroupUpdate persisted the review/QA
3453
+ // patches through the state FILE (the watcher's in-memory copy only tracks
3454
+ // worker completion), so the authoritative post-round state is on disk.
3455
+ const finalState = await loadState(statePath)
3456
+
3457
+ const terminal = finalState.groups.filter(
3458
+ (g) => g.status === "done" || g.status === "failed" || g.status === "needs-human",
3459
+ )
3460
+ process.stdout.write(
3461
+ `\nRound complete: ${terminal.length}/${finalState.groups.length} groups terminal. State: ${statePath}\n`,
3462
+ )
3463
+ const failed = finalState.groups.filter((g) => g.status === "failed")
3464
+ if (failed.length > 0) {
3465
+ process.stderr.write(
3466
+ `headlesscode orchestrate: ${failed.length} group(s) failed: ${failed.map((g) => g.name).join(", ")}\n`,
3467
+ )
3468
+ return 1
3469
+ }
3470
+ // Rework-exhausted groups are a DISTINCT, final "needs a human" outcome —
3471
+ // not a mid-review group and not a generic failure. Exit 1 like a failure
3472
+ // (the round did not fully succeed) but the message names the cause.
3473
+ const needsHuman = finalState.groups.filter((g) => g.status === "needs-human")
3474
+ if (needsHuman.length > 0) {
3475
+ process.stderr.write(
3476
+ `headlesscode orchestrate: ${needsHuman.length} group(s) need a human (rework/continuation ` +
3477
+ `attempts exhausted): ${needsHuman.map((g) => g.name).join(", ")}. ` +
3478
+ `See ${statePath} for their review_verdict/pending_review_findings and continuationCount.\n`,
3479
+ )
3480
+ return 1
3481
+ }
3482
+
3483
+ // Phase 4: human-approval deploy gate (--deploy). Runs ONLY when every
3484
+ // group is done, no group failed, every review is clean (when review is
3485
+ // enabled) and every QA verdict is pass (when --qa was passed). The gate
3486
+ // itself is a hard stop: scripts/deploy-gate.sh never invokes the repo's
3487
+ // deploy-production.sh without explicit human approval.
3488
+ if (options.deploy) {
3489
+ const gate = deployGateReady(finalState, options)
3490
+ if (!gate.ready) {
3491
+ process.stderr.write(
3492
+ `headlesscode orchestrate: deploy gate SKIPPED — ${gate.reason}. No deploy was attempted.\n`,
3493
+ )
3494
+ return 1
3495
+ }
3496
+ const batch = options.batch ?? `round-${new Date().toISOString().slice(0, 10)}`
3497
+ const gateCmd = [
3498
+ `bash ${path.join(HARNESS_ROOT_TS, "scripts", "deploy-gate.sh")}`,
3499
+ `"${repo}"`,
3500
+ `--batch "${batch}"`,
3501
+ options.deployArgs ? `--deploy-args "${options.deployArgs}"` : "",
3502
+ ]
3503
+ .filter(Boolean)
3504
+ .join(" ")
3505
+ process.stdout.write(`[orchestrate] running deploy gate...\n`)
3506
+ // Same `bash -c` pattern as the spawn call above (no double-bash).
3507
+ const gateResult = spawnSync("bash", ["-c", gateCmd], {
3508
+ cwd: repo,
3509
+ env: { ...process.env },
3510
+ stdio: "inherit",
3511
+ })
3512
+ if (gateResult.status === 3) {
3513
+ process.stderr.write(
3514
+ "headlesscode orchestrate: deploy DENIED by the human-approval gate. deploy-production.sh was NOT run.\n",
3515
+ )
3516
+ return 1
3517
+ }
3518
+ if (gateResult.status !== 0) {
3519
+ process.stderr.write(
3520
+ `headlesscode orchestrate: deploy gate failed (exit ${gateResult.status}). No deploy was attempted.\n`,
3521
+ )
3522
+ return 1
3523
+ }
3524
+ process.stdout.write("[orchestrate] deploy approved by the gate and executed.\n")
3525
+ }
3526
+ return 0
3527
+ }
3528
+
3529
+ /**
3530
+ * Phase 4: decide whether the deploy gate may run after a round. Fail-closed:
3531
+ * returns { ready: false, reason } unless every group is done, none failed,
3532
+ * every review is clean (when review is enabled) and every QA verdict is pass
3533
+ * (when --qa was passed). A group that never got a QA record (e.g. review
3534
+ * found issues) blocks the deploy.
3535
+ */
3536
+ function deployGateReady(
3537
+ state: OrchestratorState,
3538
+ options: OrchestrateOptions,
3539
+ ): { ready: boolean; reason?: string } {
3540
+ if (state.groups.length === 0) {
3541
+ return { ready: false, reason: "no groups in state" }
3542
+ }
3543
+ const anyFailed = state.groups.some((g) => g.status === "failed")
3544
+ if (anyFailed) {
3545
+ return { ready: false, reason: "one or more groups failed" }
3546
+ }
3547
+ // Rework/continuation-exhausted groups: a distinct, final "needs a human"
3548
+ // outcome — the deploy must never proceed while one exists, and the reason
3549
+ // must say why (not a generic failure, not an in-flight review).
3550
+ const anyNeedsHuman = state.groups.some((g) => g.status === "needs-human")
3551
+ if (anyNeedsHuman) {
3552
+ return { ready: false, reason: "one or more groups need human attention (rework/continuation attempts exhausted)" }
3553
+ }
3554
+ const anyNotDone = state.groups.some((g) => g.status !== "done")
3555
+ if (anyNotDone) {
3556
+ return { ready: false, reason: "not all groups reached status done" }
3557
+ }
3558
+ if (options.review) {
3559
+ const anyUnreviewed = state.groups.some((g) => g.review_verdict !== "clean")
3560
+ if (anyUnreviewed) {
3561
+ return { ready: false, reason: "not all groups reviewed clean" }
3562
+ }
3563
+ }
3564
+ if (options.qa) {
3565
+ const anyQaUnpassed = state.groups.some((g) => g.qa?.verdict !== "pass")
3566
+ if (anyQaUnpassed) {
3567
+ return { ready: false, reason: "not all groups passed QA" }
3568
+ }
3569
+ }
3570
+ return { ready: true }
3571
+ }