headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,3571 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `headlesscode orchestrate` subcommand — the thin orchestrator entry point.
|
|
3
|
+
*
|
|
4
|
+
* npx tsx src/cli.ts orchestrate --repo <path> --issue 27 --issue 29 \
|
|
5
|
+
* [--mode code] [--issues-json <path>] [--dry-run] [--no-review]
|
|
6
|
+
*
|
|
7
|
+
* Flow:
|
|
8
|
+
* 1. Read issues (`gh issue view --json number,title,body` when gh is
|
|
9
|
+
* available, else `--issues-json <file>`).
|
|
10
|
+
* 2. Split them into worktree groups with splitIssues (the
|
|
11
|
+
* multi-agent-orchestrator heuristics).
|
|
12
|
+
* 3. Generate task files under <repo>/plans/parallel-tasks/.
|
|
13
|
+
* 4. Spawn via scripts/spawn-parallel-worktrees.sh (the bash script remains
|
|
14
|
+
* the ACTUAL spawner — the CLI only assembles the spec triples).
|
|
15
|
+
* 5. Watch for completion (.harness.done markers) and run the headless
|
|
16
|
+
* reviewer on each group that finished cleanly.
|
|
17
|
+
*
|
|
18
|
+
* `--dry-run` prints the split plan + the exact spawn command without
|
|
19
|
+
* spawning anything (no API key needed).
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
import { execFileSync, spawn, spawnSync } from "node:child_process"
|
|
23
|
+
import * as fs from "node:fs"
|
|
24
|
+
import * as path from "node:path"
|
|
25
|
+
import { fileURLToPath } from "node:url"
|
|
26
|
+
|
|
27
|
+
import { activeSessionCountForRepo, maxConcurrentSessionsFromEnv } from "../budget/concurrency.js"
|
|
28
|
+
import { issueShape, issueSizeWarnings, splitIssues, type IssueSizeWarning, type SplitIssue, type WorktreeSpec } from "./split.js"
|
|
29
|
+
import { readCostHistory } from "./cost-history.js"
|
|
30
|
+
import { buildEstimateSection, estimateGroups } from "./cost-estimate.js"
|
|
31
|
+
import {
|
|
32
|
+
loadState,
|
|
33
|
+
loadStateSync,
|
|
34
|
+
mutateState,
|
|
35
|
+
patchGroup,
|
|
36
|
+
saveStateSync,
|
|
37
|
+
updateGroup,
|
|
38
|
+
type OrchestratorGroup,
|
|
39
|
+
type OrchestratorState,
|
|
40
|
+
} from "./state.js"
|
|
41
|
+
import { runReviewWithRetries, type ReviewResult } from "./reviewer.js"
|
|
42
|
+
import { hasRealVerificationActivity, hasRealWorktreeChanges } from "./verification-gate.js"
|
|
43
|
+
import { runFilingStage, runResearchStage } from "./pipeline.js"
|
|
44
|
+
import { analyzeWorktreeSessions } from "./log-analysis.js"
|
|
45
|
+
import { runQaWithRetries, type QaResult } from "../qa/qa.js"
|
|
46
|
+
import { groupWorktreePath, isIterationExhaustion, isPidAlive, isProviderFailure, watchGroups } from "./watch.js"
|
|
47
|
+
import { clearHandoffSummary, readHandoffSummary } from "../engine/handoff.js"
|
|
48
|
+
import { autoSplitOversizedIssues, proposeSemanticSplit } from "./auto-split.js"
|
|
49
|
+
import { DEFAULT_MODEL, OpenRouterClient } from "../llm/openrouter.js"
|
|
50
|
+
import {
|
|
51
|
+
buildStatusSummary,
|
|
52
|
+
formatStatusText,
|
|
53
|
+
isTerminalStatus,
|
|
54
|
+
reconcileGroups,
|
|
55
|
+
verdictLine,
|
|
56
|
+
waitForTerminalState,
|
|
57
|
+
DEFAULT_STATUS_POLL_INTERVAL_MS,
|
|
58
|
+
DEFAULT_STATUS_TIMEOUT_MS,
|
|
59
|
+
type StatusSummary,
|
|
60
|
+
} from "./status.js"
|
|
61
|
+
import { assessGroupCleanupSync, cleanupMain, resolveBaseBranch, type CleanupStatus } from "./cleanup.js"
|
|
62
|
+
import {
|
|
63
|
+
branchSyncStatus,
|
|
64
|
+
ORCHESTRATE_SYNC_DISABLED_ENV,
|
|
65
|
+
syncBranchWithOrigin,
|
|
66
|
+
syncSummaryLines,
|
|
67
|
+
syncWarningLines,
|
|
68
|
+
TRIVIAL_DRIFT_AHEAD,
|
|
69
|
+
} from "./git-sync.js"
|
|
70
|
+
import { resolveModelForMode } from "../config/mode-models.js"
|
|
71
|
+
import { runPreflight, runLocalPreflight, type PreflightResult, type LocalPreflightResult } from "../llm/preflight.js"
|
|
72
|
+
import { resolvePerModeEnv } from "../cli.js"
|
|
73
|
+
|
|
74
|
+
const ORCHESTRATE_USAGE = `headlesscode orchestrate — Phase 2 parallel round
|
|
75
|
+
|
|
76
|
+
Usage:
|
|
77
|
+
headlesscode orchestrate --repo <path> --issue <n> [--issue <n> ...] [options]
|
|
78
|
+
headlesscode orchestrate --repo <path> --issues-json <file> [options]
|
|
79
|
+
|
|
80
|
+
Options:
|
|
81
|
+
--repo <path> Target repo root (required)
|
|
82
|
+
--issue <n> Issue number to include (repeatable)
|
|
83
|
+
--issues-json <file> Read issues from a JSON array of {number,title,body}
|
|
84
|
+
--file-issues With --issues-json: file a REAL GitHub issue for each
|
|
85
|
+
synthetic entry (gh issue create --repo <origin-owner>/<origin-repo>),
|
|
86
|
+
swap in the real number returned, and print one
|
|
87
|
+
confirmation line per created issue. A real, visible
|
|
88
|
+
write to GitHub — opt-in, never automatic.
|
|
89
|
+
Requires --issues-json (nothing to file otherwise)
|
|
90
|
+
--mode <slug> Harness mode for workers (default: code)
|
|
91
|
+
--model <id> Explicit model override for workers + reviewer + QA
|
|
92
|
+
(beats .headlesscode/mode-models.json entries). Without
|
|
93
|
+
it, each role resolves its OWN model from the file:
|
|
94
|
+
worker mode / --review-mode / --qa-mode
|
|
95
|
+
(default: $OPENROUTER_MODEL or the client default)
|
|
96
|
+
--batch <name> Batch id recorded in the state file (default: round-<date>)
|
|
97
|
+
--review-mode <slug> Mode slug used for review sessions (default: deepseek-reviewer)
|
|
98
|
+
--no-review Spawn + watch only; do not run the reviewer
|
|
99
|
+
--qa Run a headless QA session (--mode qa-agent) on each group
|
|
100
|
+
after its review passes; record qa {status,verdict,evidence}
|
|
101
|
+
--qa-mode <slug> Mode slug for QA sessions (default: qa-agent; the target
|
|
102
|
+
repo's .roomodes + .roo/rules-<slug>/ are spliced automatically)
|
|
103
|
+
--deploy After all groups done + reviewed + QA passed, run the
|
|
104
|
+
human-approval deploy gate (scripts/deploy-gate.sh) which
|
|
105
|
+
refuses to run the repo's deploy-production.sh without
|
|
106
|
+
explicit human approval (interactive on a TTY, token/file
|
|
107
|
+
otherwise). Never auto-approves.
|
|
108
|
+
--deploy-args <str> Deploy args forwarded to the deploy script after the gate
|
|
109
|
+
approves (space-separated flags; also DEPLOY_ARGS env)
|
|
110
|
+
--poll-interval-ms <n> Watcher poll interval (default: 5000)
|
|
111
|
+
--memory-dir <path> Phase 3 memory dir for workers (passed as HEADLESSCODE_MEMORY_DIR,
|
|
112
|
+
which run-worker.sh forwards as --memory-dir to each worker CLI)
|
|
113
|
+
--max-concurrent-sessions <n> Phase 6 GLOBAL cap on concurrent sessions across
|
|
114
|
+
processes (default: $HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3).
|
|
115
|
+
When the cap is already reached this run ABORTS with a clear
|
|
116
|
+
message and exit 1 (never queues silently)
|
|
117
|
+
--max-rework-cycles <n> Max automatic rework attempts per group when the review
|
|
118
|
+
finds issues (default: 3). Each rework re-runs a worker on the
|
|
119
|
+
SAME worktree to fix the findings. After the cap, the group is
|
|
120
|
+
marked "needs-human" and left for manual resolution
|
|
121
|
+
--max-iterations <n> Per-session loop iteration cap for EVERY worker in this
|
|
122
|
+
round (default: $HEADLESSCODE_MAX_ITERATIONS or the harness's
|
|
123
|
+
own default, 50). A worker that hits the cap is auto-continued
|
|
124
|
+
on the SAME worktree (see --max-continuations) instead of
|
|
125
|
+
hard-failing the group
|
|
126
|
+
--max-continuations <n> Max automatic continuations per group after a worker
|
|
127
|
+
hits the iteration cap (default: 3). Each continuation
|
|
128
|
+
re-spawns a worker on the SAME worktree with a fresh session.
|
|
129
|
+
After the cap, the group is marked "needs-human" and left for
|
|
130
|
+
manual resolution
|
|
131
|
+
--plan-first Issue #49 experiment: run a SHORT architect-mode
|
|
132
|
+
planning session per worktree BEFORE the code worker, and
|
|
133
|
+
append its plan (PLAN.md) into the worker's task file so the
|
|
134
|
+
code worker executes against it instead of re-discovering
|
|
135
|
+
context (grep/read cycles) from iteration 1. OPT-IN — never
|
|
136
|
+
the default without evidence it helps. The plan session runs
|
|
137
|
+
synchronously inside the spawner, so N plan-first worktrees
|
|
138
|
+
extend the spawn call by ~N × plan-session time
|
|
139
|
+
--plan-first-mode <slug> Mode slug for the plan-first session (default:
|
|
140
|
+
architect; any mode the worker CLI accepts — built-in or
|
|
141
|
+
.roomodes)
|
|
142
|
+
--plan-first-max-iterations <n> Iteration cap for the plan-first session
|
|
143
|
+
(default: 15, or $HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS) —
|
|
144
|
+
deliberately short: the plan session must produce a plan,
|
|
145
|
+
not implement it
|
|
146
|
+
--no-preflight Skip the pre-spawn preflight probe (issue #13): a cheap
|
|
147
|
+
1-token completion using the EXACT model + provider pin a real
|
|
148
|
+
worker uses, run BEFORE anything is spawned, that distinguishes
|
|
149
|
+
key-invalid / account-balance-exhausted / pinned-provider-down
|
|
150
|
+
/ all-clear. ON by default; skip for CI/non-interactive
|
|
151
|
+
contexts that don't want the extra round-trip
|
|
152
|
+
--no-issue-size-check Skip the pre-flight issue-size warning (issue #53): a
|
|
153
|
+
FREE deterministic scan of each issue body that warns loudly
|
|
154
|
+
when it reads like 3+ independent pieces of work (top-level
|
|
155
|
+
numbered/bulleted sections), before anything is spawned — the
|
|
156
|
+
shape that burned iteration caps + budget on real rounds
|
|
157
|
+
. ON by default; the
|
|
158
|
+
warning never aborts the round, so this only silences it
|
|
159
|
+
--no-auto-split Skip auto-split: when an issue is flagged by the size
|
|
160
|
+
check above, an LLM proposes the smallest number of
|
|
161
|
+
independently-shippable sub-issues (grouped by
|
|
162
|
+
deliverable, NOT one per bullet — sequential steps of
|
|
163
|
+
one fix stay together), files them as real GitHub
|
|
164
|
+
issues, and closes the oversized parent with a link
|
|
165
|
+
to each. Fails open (LLM/filing error, or the model
|
|
166
|
+
deciding the issue is genuinely one coherent piece of
|
|
167
|
+
work) by dispatching the original issue unchanged.
|
|
168
|
+
ON by default when a GitHub origin remote is present
|
|
169
|
+
and issueSizeCheck is on; this flag falls back to
|
|
170
|
+
warn-only
|
|
171
|
+
--dry-run Print the split plan, a cost estimate (from recorded
|
|
172
|
+
cost-history, keyed by issue shape — issue #16), and
|
|
173
|
+
the spawn commands; spawn nothing
|
|
174
|
+
--help Show this help and exit
|
|
175
|
+
|
|
176
|
+
Environment:
|
|
177
|
+
HEADLESSCODE_OPENROUTER_API_KEY Required for a real run (workers + reviewer + QA)
|
|
178
|
+
ORCHESTRATOR_MODE Overrides --mode when set
|
|
179
|
+
HEADLESSCODE_CLI Overrides the spawner's CLI command
|
|
180
|
+
HEADLESSCODE_MEMORY_DIR Memory dir inherited by workers (unless --memory-dir overrides)
|
|
181
|
+
DEPLOY_APPROVAL_TOKEN Deploy gate token (non-interactive approval; must match the
|
|
182
|
+
token in <repo>/.deploy-approval)
|
|
183
|
+
DEPLOY_APPROVAL_FILE Path to a one-time approval file created by a human
|
|
184
|
+
(default: <repo>/.worktrees/.deploy-approved-<batch>)
|
|
185
|
+
DEPLOY_ARGS Deploy args forwarded to the deploy script (see --deploy-args)
|
|
186
|
+
HEADLESSCODE_MAX_CONCURRENT_SESSIONS Global concurrency cap (default 3)
|
|
187
|
+
HEADLESSCODE_MAX_COST_USD / HEADLESSCODE_MAX_DURATION_MS
|
|
188
|
+
Per-session budget env fallbacks forwarded to workers
|
|
189
|
+
(run-worker.sh / run-qa.sh pass them as CLI flags)
|
|
190
|
+
HEADLESSCODE_MAX_ITERATIONS
|
|
191
|
+
Per-session iteration cap env fallback forwarded to
|
|
192
|
+
workers as --max-iterations (run-worker.sh)
|
|
193
|
+
`
|
|
194
|
+
|
|
195
|
+
interface OrchestrateOptions {
|
|
196
|
+
repo: string
|
|
197
|
+
issues: number[]
|
|
198
|
+
issuesJson?: string
|
|
199
|
+
/**
|
|
200
|
+
* Fix 3: file a REAL GitHub issue for every synthetic --issues-json entry
|
|
201
|
+
* and substitute the real number before anything else touches `issues`.
|
|
202
|
+
* Opt-in — a real write to a real GitHub repo. Requires --issues-json.
|
|
203
|
+
*/
|
|
204
|
+
fileIssues: boolean
|
|
205
|
+
mode: string
|
|
206
|
+
model?: string
|
|
207
|
+
batch?: string
|
|
208
|
+
reviewMode: string
|
|
209
|
+
review: boolean
|
|
210
|
+
/** Phase 4: run a headless QA session per group after review passes. */
|
|
211
|
+
qa: boolean
|
|
212
|
+
/** Phase 4: mode slug for QA sessions (default qa-agent). */
|
|
213
|
+
qaMode: string
|
|
214
|
+
/** Phase 4: run the human-approval deploy gate after everything passes. */
|
|
215
|
+
deploy: boolean
|
|
216
|
+
/** Phase 4: deploy args forwarded to the deploy script after approval. */
|
|
217
|
+
deployArgs?: string
|
|
218
|
+
pollIntervalMs: number
|
|
219
|
+
memoryDir?: string
|
|
220
|
+
/** Phase 6: global concurrent-session cap (default env or 3). */
|
|
221
|
+
maxConcurrentSessions: number
|
|
222
|
+
/** Rework loop: max review-rework attempts per group before giving up (default 3). */
|
|
223
|
+
maxReworkCycles: number
|
|
224
|
+
/**
|
|
225
|
+
* Per-session iteration cap forwarded to every worker (default:
|
|
226
|
+
* $HEADLESSCODE_MAX_ITERATIONS; undefined = leave the harness's own
|
|
227
|
+
* default, 50, untouched).
|
|
228
|
+
*/
|
|
229
|
+
maxIterations?: number
|
|
230
|
+
/** Auto-continue: max re-spawns per group after a worker hits the iteration cap (default 3). */
|
|
231
|
+
maxContinuations: number
|
|
232
|
+
/**
|
|
233
|
+
* Issue #49 experiment: run a short architect-mode planning session per
|
|
234
|
+
* worktree BEFORE the code worker, and append its plan (PLAN.md) into the
|
|
235
|
+
* worker's task file so the code worker executes against it instead of
|
|
236
|
+
* re-discovering context from iteration 1. OPT-IN — never the default.
|
|
237
|
+
*/
|
|
238
|
+
planFirst: boolean
|
|
239
|
+
/** Mode slug for the plan-first session (default: architect). */
|
|
240
|
+
planFirstMode: string
|
|
241
|
+
/** Iteration cap for the plan-first session (default: 15 — deliberately short). */
|
|
242
|
+
planFirstMaxIterations: number
|
|
243
|
+
/**
|
|
244
|
+
* Issue #13 pre-spawn preflight probe (1-token, exact model + provider
|
|
245
|
+
* pin). ON by default; --no-preflight skips it for CI/non-interactive.
|
|
246
|
+
*/
|
|
247
|
+
preflight: boolean
|
|
248
|
+
/**
|
|
249
|
+
* Issue #53 pre-flight issue-size check: a FREE deterministic scan that
|
|
250
|
+
* warns loudly (never aborts) when an issue body reads like 3+ independent
|
|
251
|
+
* pieces of work, before anything is spawned. ON by default;
|
|
252
|
+
* --no-issue-size-check silences it.
|
|
253
|
+
*/
|
|
254
|
+
issueSizeCheck: boolean
|
|
255
|
+
/**
|
|
256
|
+
* Auto-split (issue #53 follow-up): when the size check flags an issue,
|
|
257
|
+
* propose a SEMANTIC split (LLM-decided independent deliverables, not a
|
|
258
|
+
* mechanical one-sub-issue-per-bullet explosion — see auto-split.ts) and
|
|
259
|
+
* file the result as real GitHub issues, closing the oversized parent.
|
|
260
|
+
* ON by default when a GitHub 'origin' remote is available; --no-auto-split
|
|
261
|
+
* falls back to warn-only. Requires issueSizeCheck (nothing to act on
|
|
262
|
+
* otherwise) and a real (non---issues-json) round (there is no GitHub
|
|
263
|
+
* issue to close for a synthetic entry).
|
|
264
|
+
*/
|
|
265
|
+
autoSplit: boolean
|
|
266
|
+
dryRun: boolean
|
|
267
|
+
help: boolean
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
/**
|
|
271
|
+
* Resolve the default per-session iteration cap from $HEADLESSCODE_MAX_ITERATIONS.
|
|
272
|
+
* undefined (unset or invalid) means "leave the harness's own default (50)
|
|
273
|
+
* untouched" — the harness default itself is never changed, this only makes the
|
|
274
|
+
* round-level cap overridable.
|
|
275
|
+
*/
|
|
276
|
+
function maxIterationsFromEnv(): number | undefined {
|
|
277
|
+
const raw = process.env.HEADLESSCODE_MAX_ITERATIONS
|
|
278
|
+
if (raw === undefined || raw === "") {
|
|
279
|
+
return undefined
|
|
280
|
+
}
|
|
281
|
+
const n = Number(raw)
|
|
282
|
+
return Number.isInteger(n) && n > 0 ? n : undefined
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
/**
|
|
286
|
+
* Resolve the plan-first session's iteration cap from
|
|
287
|
+
* $HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS (default: 15 — the plan session is
|
|
288
|
+
* deliberately SHORT: it must produce a plan, not implement it). Invalid
|
|
289
|
+
* values fall back to the default.
|
|
290
|
+
*/
|
|
291
|
+
function planFirstMaxIterationsFromEnv(): number {
|
|
292
|
+
const raw = process.env.HEADLESSCODE_PLAN_FIRST_MAX_ITERATIONS
|
|
293
|
+
if (raw === undefined || raw === "") {
|
|
294
|
+
return 15
|
|
295
|
+
}
|
|
296
|
+
const n = Number(raw)
|
|
297
|
+
return Number.isInteger(n) && n > 0 ? n : 15
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
export function parseOrchestrateArgs(argv: string[]): { options: OrchestrateOptions; error?: string } {
|
|
301
|
+
const options: OrchestrateOptions = {
|
|
302
|
+
repo: "",
|
|
303
|
+
issues: [],
|
|
304
|
+
mode: process.env.ORCHESTRATOR_MODE ?? "code",
|
|
305
|
+
reviewMode: "deepseek-reviewer",
|
|
306
|
+
review: true,
|
|
307
|
+
qa: false,
|
|
308
|
+
qaMode: "qa-agent",
|
|
309
|
+
deploy: false,
|
|
310
|
+
fileIssues: false,
|
|
311
|
+
pollIntervalMs: 5000,
|
|
312
|
+
maxConcurrentSessions: maxConcurrentSessionsFromEnv(),
|
|
313
|
+
maxReworkCycles: 3,
|
|
314
|
+
maxContinuations: 3,
|
|
315
|
+
maxIterations: maxIterationsFromEnv(),
|
|
316
|
+
planFirst: false,
|
|
317
|
+
planFirstMode: process.env.HEADLESSCODE_PLAN_FIRST_MODE ?? "architect",
|
|
318
|
+
planFirstMaxIterations: planFirstMaxIterationsFromEnv(),
|
|
319
|
+
preflight: true,
|
|
320
|
+
issueSizeCheck: true,
|
|
321
|
+
autoSplit: true,
|
|
322
|
+
dryRun: false,
|
|
323
|
+
help: false,
|
|
324
|
+
}
|
|
325
|
+
|
|
326
|
+
for (let i = 0; i < argv.length; i++) {
|
|
327
|
+
const arg = argv[i]
|
|
328
|
+
const eq = arg.indexOf("=")
|
|
329
|
+
const flag = eq === -1 ? arg : arg.slice(0, eq)
|
|
330
|
+
const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
|
|
331
|
+
const next = (): string | undefined => {
|
|
332
|
+
if (inlineValue !== undefined) {
|
|
333
|
+
return inlineValue
|
|
334
|
+
}
|
|
335
|
+
const v = argv[i + 1]
|
|
336
|
+
if (v === undefined || v.startsWith("--")) {
|
|
337
|
+
return undefined
|
|
338
|
+
}
|
|
339
|
+
i++
|
|
340
|
+
return v
|
|
341
|
+
}
|
|
342
|
+
|
|
343
|
+
switch (flag) {
|
|
344
|
+
case "--repo": {
|
|
345
|
+
const v = next()
|
|
346
|
+
if (v === undefined) {
|
|
347
|
+
return { options, error: "Missing value for --repo" }
|
|
348
|
+
}
|
|
349
|
+
options.repo = v
|
|
350
|
+
break
|
|
351
|
+
}
|
|
352
|
+
case "--issue": {
|
|
353
|
+
const v = next()
|
|
354
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
355
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
356
|
+
return { options, error: "--issue requires a positive integer" }
|
|
357
|
+
}
|
|
358
|
+
options.issues.push(n)
|
|
359
|
+
break
|
|
360
|
+
}
|
|
361
|
+
case "--issues-json": {
|
|
362
|
+
const v = next()
|
|
363
|
+
if (v === undefined) {
|
|
364
|
+
return { options, error: "Missing value for --issues-json" }
|
|
365
|
+
}
|
|
366
|
+
options.issuesJson = v
|
|
367
|
+
break
|
|
368
|
+
}
|
|
369
|
+
case "--file-issues":
|
|
370
|
+
options.fileIssues = true
|
|
371
|
+
break
|
|
372
|
+
case "--mode": {
|
|
373
|
+
const v = next()
|
|
374
|
+
if (v === undefined) {
|
|
375
|
+
return { options, error: "Missing value for --mode" }
|
|
376
|
+
}
|
|
377
|
+
options.mode = v
|
|
378
|
+
break
|
|
379
|
+
}
|
|
380
|
+
case "--model": {
|
|
381
|
+
const v = next()
|
|
382
|
+
if (v === undefined) {
|
|
383
|
+
return { options, error: "Missing value for --model" }
|
|
384
|
+
}
|
|
385
|
+
options.model = v
|
|
386
|
+
break
|
|
387
|
+
}
|
|
388
|
+
case "--batch": {
|
|
389
|
+
const v = next()
|
|
390
|
+
if (v === undefined) {
|
|
391
|
+
return { options, error: "Missing value for --batch" }
|
|
392
|
+
}
|
|
393
|
+
options.batch = v
|
|
394
|
+
break
|
|
395
|
+
}
|
|
396
|
+
case "--review-mode": {
|
|
397
|
+
const v = next()
|
|
398
|
+
if (v === undefined) {
|
|
399
|
+
return { options, error: "Missing value for --review-mode" }
|
|
400
|
+
}
|
|
401
|
+
options.reviewMode = v
|
|
402
|
+
break
|
|
403
|
+
}
|
|
404
|
+
case "--poll-interval-ms": {
|
|
405
|
+
const v = next()
|
|
406
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
407
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
408
|
+
return { options, error: "--poll-interval-ms requires a positive integer" }
|
|
409
|
+
}
|
|
410
|
+
options.pollIntervalMs = n
|
|
411
|
+
break
|
|
412
|
+
}
|
|
413
|
+
case "--memory-dir": {
|
|
414
|
+
const v = next()
|
|
415
|
+
if (v === undefined) {
|
|
416
|
+
return { options, error: "Missing value for --memory-dir" }
|
|
417
|
+
}
|
|
418
|
+
options.memoryDir = v
|
|
419
|
+
break
|
|
420
|
+
}
|
|
421
|
+
case "--max-concurrent-sessions": {
|
|
422
|
+
const v = next()
|
|
423
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
424
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
425
|
+
return { options, error: "--max-concurrent-sessions requires a positive integer" }
|
|
426
|
+
}
|
|
427
|
+
options.maxConcurrentSessions = n
|
|
428
|
+
break
|
|
429
|
+
}
|
|
430
|
+
case "--max-rework-cycles": {
|
|
431
|
+
const v = next()
|
|
432
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
433
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
434
|
+
return { options, error: "--max-rework-cycles requires a positive integer" }
|
|
435
|
+
}
|
|
436
|
+
options.maxReworkCycles = n
|
|
437
|
+
break
|
|
438
|
+
}
|
|
439
|
+
case "--max-iterations": {
|
|
440
|
+
const v = next()
|
|
441
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
442
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
443
|
+
return { options, error: "--max-iterations requires a positive integer" }
|
|
444
|
+
}
|
|
445
|
+
options.maxIterations = n
|
|
446
|
+
break
|
|
447
|
+
}
|
|
448
|
+
case "--max-continuations": {
|
|
449
|
+
const v = next()
|
|
450
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
451
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
452
|
+
return { options, error: "--max-continuations requires a positive integer" }
|
|
453
|
+
}
|
|
454
|
+
options.maxContinuations = n
|
|
455
|
+
break
|
|
456
|
+
}
|
|
457
|
+
case "--plan-first":
|
|
458
|
+
options.planFirst = true
|
|
459
|
+
break
|
|
460
|
+
case "--plan-first-mode": {
|
|
461
|
+
const v = next()
|
|
462
|
+
if (v === undefined) {
|
|
463
|
+
return { options, error: "Missing value for --plan-first-mode" }
|
|
464
|
+
}
|
|
465
|
+
options.planFirstMode = v
|
|
466
|
+
break
|
|
467
|
+
}
|
|
468
|
+
case "--plan-first-max-iterations": {
|
|
469
|
+
const v = next()
|
|
470
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
471
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
472
|
+
return { options, error: "--plan-first-max-iterations requires a positive integer" }
|
|
473
|
+
}
|
|
474
|
+
options.planFirstMaxIterations = n
|
|
475
|
+
break
|
|
476
|
+
}
|
|
477
|
+
case "--no-review":
|
|
478
|
+
options.review = false
|
|
479
|
+
break
|
|
480
|
+
case "--qa":
|
|
481
|
+
options.qa = true
|
|
482
|
+
break
|
|
483
|
+
case "--qa-mode": {
|
|
484
|
+
const v = next()
|
|
485
|
+
if (v === undefined) {
|
|
486
|
+
return { options, error: "Missing value for --qa-mode" }
|
|
487
|
+
}
|
|
488
|
+
options.qaMode = v
|
|
489
|
+
break
|
|
490
|
+
}
|
|
491
|
+
case "--deploy":
|
|
492
|
+
options.deploy = true
|
|
493
|
+
break
|
|
494
|
+
case "--deploy-args": {
|
|
495
|
+
// Deploy args legitimately start with "--" (e.g. --local), so
|
|
496
|
+
// bypass next()'s "--"-rejecting value reader.
|
|
497
|
+
const v = inlineValue ?? argv[i + 1]
|
|
498
|
+
if (v === undefined) {
|
|
499
|
+
return { options, error: "Missing value for --deploy-args" }
|
|
500
|
+
}
|
|
501
|
+
i++
|
|
502
|
+
options.deployArgs = v
|
|
503
|
+
break
|
|
504
|
+
}
|
|
505
|
+
case "--no-preflight":
|
|
506
|
+
options.preflight = false
|
|
507
|
+
break
|
|
508
|
+
case "--no-issue-size-check":
|
|
509
|
+
options.issueSizeCheck = false
|
|
510
|
+
break
|
|
511
|
+
case "--no-auto-split":
|
|
512
|
+
options.autoSplit = false
|
|
513
|
+
break
|
|
514
|
+
case "--dry-run":
|
|
515
|
+
options.dryRun = true
|
|
516
|
+
break
|
|
517
|
+
case "--help":
|
|
518
|
+
case "-h":
|
|
519
|
+
options.help = true
|
|
520
|
+
break
|
|
521
|
+
default:
|
|
522
|
+
return { options, error: `Unknown orchestrate argument: ${arg}` }
|
|
523
|
+
}
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
// Fix 3: --file-issues without --issues-json is a usage error — a real
|
|
527
|
+
// --issue <n>/gh-sourced entry already carries a real number, so there is
|
|
528
|
+
// nothing to file. (Match the existing usage-error exit-2 convention.)
|
|
529
|
+
if (options.fileIssues && !options.issuesJson) {
|
|
530
|
+
return { options, error: "--file-issues requires --issues-json (a real --issue <n> round has nothing to file)" }
|
|
531
|
+
}
|
|
532
|
+
|
|
533
|
+
return { options }
|
|
534
|
+
}
|
|
535
|
+
|
|
536
|
+
// ─── Issue loading ───────────────────────────────────────────────────────────
|
|
537
|
+
|
|
538
|
+
function ghAvailable(): boolean {
|
|
539
|
+
try {
|
|
540
|
+
execFileSync("gh", ["--version"], { stdio: "ignore", timeout: 5000 })
|
|
541
|
+
return true
|
|
542
|
+
} catch {
|
|
543
|
+
return false
|
|
544
|
+
}
|
|
545
|
+
}
|
|
546
|
+
|
|
547
|
+
function fetchIssueWithGh(repo: string, number: number): SplitIssue {
|
|
548
|
+
const out = execFileSync("gh", ["issue", "view", String(number), "--json", "number,title,body"], {
|
|
549
|
+
cwd: repo,
|
|
550
|
+
encoding: "utf-8",
|
|
551
|
+
timeout: 30_000,
|
|
552
|
+
})
|
|
553
|
+
const parsed = JSON.parse(out) as { number?: number; title?: string; body?: string | null }
|
|
554
|
+
return { number: parsed.number ?? number, title: parsed.title ?? `issue ${number}`, body: parsed.body ?? undefined }
|
|
555
|
+
}
|
|
556
|
+
|
|
557
|
+
/**
|
|
558
|
+
* Resolve `<owner>/<repo>` from the target repo's `origin` remote (used by
|
|
559
|
+
* --file-issues' `gh issue create --repo <owner>/<repo>`). Accepts both ssh
|
|
560
|
+
* (`git@host:owner/repo.git`) and https (`https://host/owner/repo.git`) URL
|
|
561
|
+
* forms. Returns undefined when there is no origin or it is unparseable.
|
|
562
|
+
*/
|
|
563
|
+
function originOwnerRepo(repo: string): string | undefined {
|
|
564
|
+
try {
|
|
565
|
+
const url = execFileSync("git", ["-C", repo, "remote", "get-url", "origin"], {
|
|
566
|
+
encoding: "utf-8",
|
|
567
|
+
timeout: 10_000,
|
|
568
|
+
}).trim()
|
|
569
|
+
const match = url.match(/^(?:git@[^:]+:|https?:\/\/[^/]+\/)([^/]+)\/([^/]+?)(?:\.git)?$/)
|
|
570
|
+
return match ? `${match[1]}/${match[2]}` : undefined
|
|
571
|
+
} catch {
|
|
572
|
+
return undefined
|
|
573
|
+
}
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
/**
|
|
577
|
+
* File one real GitHub issue via `gh issue create` — mirrors fetchIssueWithGh's
|
|
578
|
+
* execFileSync idiom (same cwd/timeout). Returns the real issue number (the
|
|
579
|
+
* trailing number of the issue URL gh prints — parsed rather than depending on
|
|
580
|
+
* a --json flag that only exists in newer gh versions) and its URL.
|
|
581
|
+
*/
|
|
582
|
+
function createGhIssue(repo: string, ownerRepo: string, issue: SplitIssue): { number: number; url: string } {
|
|
583
|
+
const out = execFileSync(
|
|
584
|
+
"gh",
|
|
585
|
+
["issue", "create", "--repo", ownerRepo, "--title", issue.title, "--body", issue.body ?? ""],
|
|
586
|
+
{ cwd: repo, encoding: "utf-8", timeout: 30_000 },
|
|
587
|
+
)
|
|
588
|
+
const url = out.trim().split(/\s+/).pop() ?? out.trim()
|
|
589
|
+
const numberMatch = url.match(/\/issues\/(\d+)\/?$/)
|
|
590
|
+
if (!numberMatch) {
|
|
591
|
+
throw new Error(`gh issue create returned an unrecognized response: ${out.trim()}`)
|
|
592
|
+
}
|
|
593
|
+
return { number: Number(numberMatch[1]), url }
|
|
594
|
+
}
|
|
595
|
+
|
|
596
|
+
/**
|
|
597
|
+
* Close a real GitHub issue with a comment (auto-split's parent-issue
|
|
598
|
+
* closeout) — mirrors createGhIssue's execFileSync idiom. `gh issue close
|
|
599
|
+
* --comment` posts the comment atomically with the close, so there's no
|
|
600
|
+
* window where the issue is closed without the sub-issue links, or vice
|
|
601
|
+
* versa.
|
|
602
|
+
*/
|
|
603
|
+
function closeGhIssue(repo: string, ownerRepo: string, issueNumber: number, comment: string): void {
|
|
604
|
+
execFileSync("gh", ["issue", "close", String(issueNumber), "--repo", ownerRepo, "--comment", comment], {
|
|
605
|
+
cwd: repo,
|
|
606
|
+
encoding: "utf-8",
|
|
607
|
+
timeout: 30_000,
|
|
608
|
+
})
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
/**
|
|
612
|
+
* Fix 3: file a real GitHub issue for each synthetic issue and substitute the
|
|
613
|
+
* real number, so every downstream step (worktree/branch naming, task files,
|
|
614
|
+
* state persistence, PR-closing comments) uses a number that actually exists
|
|
615
|
+
* on GitHub — no other code needs to know issues were just filed. `shouldFile`
|
|
616
|
+
* gates which entries get filed: orchestrate uses the default (file all —
|
|
617
|
+
* --file-issues requires --issues-json, and every --issues-json entry is
|
|
618
|
+
* synthetic by definition); the predicate exists so tests can cover "entries
|
|
619
|
+
* that already carry a real number pass through untouched" without a real gh
|
|
620
|
+
* call. `createIssue` is injected for the same reason (real: createGhIssue).
|
|
621
|
+
*/
|
|
622
|
+
export function fileSyntheticIssues(
|
|
623
|
+
issues: SplitIssue[],
|
|
624
|
+
createIssue: (issue: SplitIssue) => { number: number; url: string },
|
|
625
|
+
shouldFile: (issue: SplitIssue) => boolean = () => true,
|
|
626
|
+
): { issues: SplitIssue[]; created: Array<{ number: number; title: string; url: string }> } {
|
|
627
|
+
const created: Array<{ number: number; title: string; url: string }> = []
|
|
628
|
+
const next = issues.map((issue) => {
|
|
629
|
+
if (!shouldFile(issue)) {
|
|
630
|
+
return issue
|
|
631
|
+
}
|
|
632
|
+
const real = createIssue(issue)
|
|
633
|
+
created.push({ number: real.number, title: issue.title, url: real.url })
|
|
634
|
+
return { ...issue, number: real.number }
|
|
635
|
+
})
|
|
636
|
+
return { issues: next, created }
|
|
637
|
+
}
|
|
638
|
+
|
|
639
|
+
function loadIssues(options: OrchestrateOptions): SplitIssue[] {
|
|
640
|
+
if (options.issuesJson) {
|
|
641
|
+
const raw = fs.readFileSync(path.resolve(options.issuesJson), "utf-8")
|
|
642
|
+
const parsed = JSON.parse(raw) as unknown
|
|
643
|
+
if (!Array.isArray(parsed)) {
|
|
644
|
+
throw new Error(`--issues-json must be a JSON array of {number,title,body}`)
|
|
645
|
+
}
|
|
646
|
+
return parsed.map((i) => {
|
|
647
|
+
const issue = i as { number?: unknown; title?: unknown; body?: unknown }
|
|
648
|
+
return {
|
|
649
|
+
number: Number(issue.number),
|
|
650
|
+
title: String(issue.title ?? ""),
|
|
651
|
+
body: typeof issue.body === "string" ? issue.body : undefined,
|
|
652
|
+
}
|
|
653
|
+
})
|
|
654
|
+
}
|
|
655
|
+
if (options.issues.length === 0) {
|
|
656
|
+
throw new Error("provide at least one --issue <n> or an --issues-json file")
|
|
657
|
+
}
|
|
658
|
+
if (!ghAvailable()) {
|
|
659
|
+
throw new Error(
|
|
660
|
+
"gh CLI is not available and no --issues-json was given — install gh or pass --issues-json <file>",
|
|
661
|
+
)
|
|
662
|
+
}
|
|
663
|
+
return [...new Set(options.issues)].sort((a, b) => a - b).map((n) => fetchIssueWithGh(options.repo, n))
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
// ─── Task-file generation ────────────────────────────────────────────────────
|
|
667
|
+
|
|
668
|
+
/**
|
|
669
|
+
* Rules splicing happens automatically: addCustomInstructions() (vendored,
|
|
670
|
+
* src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts) reads
|
|
671
|
+
* the central store's `rules/` + `rules-<mode>/` tiers (and the project's
|
|
672
|
+
* `.roo/` equivalents) into the system prompt before the first turn. Task
|
|
673
|
+
* files must NOT tell the model to go read a rules file itself — on a project
|
|
674
|
+
* with no project-local `.roo/rules-code/` (e.g. a fresh `headlesscode init`
|
|
675
|
+
* setup) that read fails outright and burns a consecutive-mistake budget
|
|
676
|
+
* before real work starts. Shared by all five task-file builders below.
|
|
677
|
+
*/
|
|
678
|
+
const RULES_SPLICED_NOTE = [
|
|
679
|
+
"Project and mode-specific rules (from the central shared store and this project's own",
|
|
680
|
+
".roo/rules*) are already spliced into your system prompt automatically — you do not",
|
|
681
|
+
"need to read any rules file yourself.",
|
|
682
|
+
]
|
|
683
|
+
|
|
684
|
+
/**
|
|
685
|
+
* One issue's title/body keyed by issue number as a string (JSON object keys
|
|
686
|
+
* are always strings). Persisted on OrchestratorGroup at dispatch time so the
|
|
687
|
+
* rework/QA/continuation task files — generated LATER from state alone, when
|
|
688
|
+
* the original issue objects are no longer in hand — can embed the real
|
|
689
|
+
* title/body instead of telling the worker to `gh issue view <n>` (which
|
|
690
|
+
* fails outright for synthetic --issues-json numbers).
|
|
691
|
+
*/
|
|
692
|
+
export type IssueBodies = Record<string, { title: string; body?: string }>
|
|
693
|
+
|
|
694
|
+
/**
|
|
695
|
+
* Render each assigned issue as an inline markdown block
|
|
696
|
+
* ("### Issue #<n>: <title>\n\n<body>") for a task file — the issue title/body
|
|
697
|
+
* is already in hand, so nothing needs fetching at session start. An issue
|
|
698
|
+
* whose body was never recorded gets an explicit note; an issue with NO entry
|
|
699
|
+
* at all falls back to the OLD `gh issue view <n>` instruction for that issue
|
|
700
|
+
* only (state written before bodies were persisted — never silently produce an
|
|
701
|
+
* empty assignment section).
|
|
702
|
+
*/
|
|
703
|
+
function issueBodyBlocks(
|
|
704
|
+
numbers: number[],
|
|
705
|
+
lookup: (n: number) => { title: string; body?: string } | undefined,
|
|
706
|
+
): string[] {
|
|
707
|
+
const blocks: string[] = []
|
|
708
|
+
for (const n of numbers) {
|
|
709
|
+
const entry = lookup(n)
|
|
710
|
+
if (entry === undefined) {
|
|
711
|
+
blocks.push(
|
|
712
|
+
`### Issue #${n}`,
|
|
713
|
+
"",
|
|
714
|
+
`No issue body recorded — fetch it yourself: \`gh issue view ${n}\`.`,
|
|
715
|
+
)
|
|
716
|
+
} else {
|
|
717
|
+
blocks.push(
|
|
718
|
+
`### Issue #${n}: ${entry.title}`,
|
|
719
|
+
"",
|
|
720
|
+
entry.body && entry.body.trim() !== ""
|
|
721
|
+
? entry.body
|
|
722
|
+
: "(no issue body recorded — inspect the codebase and git history for context)",
|
|
723
|
+
)
|
|
724
|
+
}
|
|
725
|
+
}
|
|
726
|
+
return blocks
|
|
727
|
+
}
|
|
728
|
+
|
|
729
|
+
export function buildTaskFileContent(spec: WorktreeSpec, issues: SplitIssue[]): string {
|
|
730
|
+
const assigned = spec.issues.map((n) => `#${n}`).join(", ")
|
|
731
|
+
const issueBlocks = issueBodyBlocks(spec.issues, (n) => issues.find((i) => i.number === n))
|
|
732
|
+
return [
|
|
733
|
+
...RULES_SPLICED_NOTE,
|
|
734
|
+
"",
|
|
735
|
+
"You are working autonomously in an isolated git worktree, on your own branch, with your own",
|
|
736
|
+
"isolated environment. No human is in the loop until you open a PR. Read this entire file before",
|
|
737
|
+
"doing anything.",
|
|
738
|
+
"",
|
|
739
|
+
"## Your assignment",
|
|
740
|
+
"",
|
|
741
|
+
`GitHub issues assigned to this worktree: ${assigned}. The real title/body of each is inline below.`,
|
|
742
|
+
"",
|
|
743
|
+
...(issueBlocks.length > 0 ? issueBlocks : ["(no issue numbers recorded — inspect the worktree for context)"]),
|
|
744
|
+
"",
|
|
745
|
+
"## Environment",
|
|
746
|
+
"",
|
|
747
|
+
`Workspace root: this worktree (branch: current branch). All file/command tools are scoped to`,
|
|
748
|
+
"this worktree. Commit per logical unit with the issue number in the message.",
|
|
749
|
+
"",
|
|
750
|
+
"## Scratch",
|
|
751
|
+
"",
|
|
752
|
+
"Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
|
|
753
|
+
"`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
|
|
754
|
+
"worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
|
|
755
|
+
"",
|
|
756
|
+
"## Workflow",
|
|
757
|
+
"",
|
|
758
|
+
"1. Read each assigned issue's inline body above and the relevant source files.",
|
|
759
|
+
"2. Implement the change following the project's own conventions and the rules files.",
|
|
760
|
+
"3. Verify for real (run the project's tests / boot check) and capture the output.",
|
|
761
|
+
"4. Commit with the issue number referenced, one logical change per commit.",
|
|
762
|
+
"5. Push the branch and open a PR (ready, not draft).",
|
|
763
|
+
"6. Post the closing comment below on each issue, then close it.",
|
|
764
|
+
"",
|
|
765
|
+
"## Required comment template",
|
|
766
|
+
"",
|
|
767
|
+
"Your issue-closing comment must use this exact structure (every section present; a section can",
|
|
768
|
+
"be 'N/A' with a reason, but a missing section is itself a review finding):",
|
|
769
|
+
"```",
|
|
770
|
+
"## Issue <n> — Closing Report",
|
|
771
|
+
"",
|
|
772
|
+
"### Files changed",
|
|
773
|
+
"<one line per file>",
|
|
774
|
+
"",
|
|
775
|
+
"### Baseline (before)",
|
|
776
|
+
"<real output>",
|
|
777
|
+
"",
|
|
778
|
+
"### Baseline (after)",
|
|
779
|
+
"<real output>",
|
|
780
|
+
"",
|
|
781
|
+
"### Boot check",
|
|
782
|
+
"<real output or exact reason it didn't apply>",
|
|
783
|
+
"",
|
|
784
|
+
"### Residual risk / notes",
|
|
785
|
+
"<anything not 100% certain>",
|
|
786
|
+
"",
|
|
787
|
+
"PR: <link>",
|
|
788
|
+
"```",
|
|
789
|
+
"",
|
|
790
|
+
"## Closing out",
|
|
791
|
+
"",
|
|
792
|
+
"Once every assigned issue is closed with the template above:",
|
|
793
|
+
"1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
|
|
794
|
+
" numbers, the boot check, and the PR link(s).",
|
|
795
|
+
"2. The harness records your completion automatically (exit code + .harness.done marker) —",
|
|
796
|
+
" no separate phone-home step is needed.",
|
|
797
|
+
"3. Do not start on any issue not assigned to this worktree.",
|
|
798
|
+
"",
|
|
799
|
+
`Scope: ${assigned} only.`,
|
|
800
|
+
].join("\n")
|
|
801
|
+
}
|
|
802
|
+
|
|
803
|
+
/**
|
|
804
|
+
* Issue #49 plan-first task content: the task file for the SHORT architect-mode
|
|
805
|
+
* planning session that runs INSIDE the fresh worktree before the code worker.
|
|
806
|
+
* The planner must produce a compact implementation plan (written to PLAN.md at
|
|
807
|
+
* the workspace root, overwriting any previous plan) and stop — never
|
|
808
|
+
* implement, never edit source, never ask the human, never switch modes. The
|
|
809
|
+
* spawner appends PLAN.md into ORCHESTRATOR_TASK.md so the code worker that
|
|
810
|
+
* follows executes against it. Reuses the other task files' conventions (rules
|
|
811
|
+
* preamble, scope line).
|
|
812
|
+
*/
|
|
813
|
+
export function buildPlanFirstTaskFileContent(
|
|
814
|
+
group: { name: string; issues?: number[] },
|
|
815
|
+
/** The in-hand issues (their title/body is embedded inline). Optional for
|
|
816
|
+
* backward compatibility with callers that only have persisted state. */
|
|
817
|
+
issues?: SplitIssue[],
|
|
818
|
+
): string {
|
|
819
|
+
const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
|
|
820
|
+
const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => issues?.find((i) => i.number === n))
|
|
821
|
+
return [
|
|
822
|
+
...RULES_SPLICED_NOTE,
|
|
823
|
+
"",
|
|
824
|
+
"You are the PLANNING phase of an autonomous two-phase workflow, running in an isolated git",
|
|
825
|
+
"worktree on your own branch. Your ONLY job is to produce a compact, actionable implementation",
|
|
826
|
+
"plan for the code session that follows. Read this entire file before doing anything.",
|
|
827
|
+
"",
|
|
828
|
+
"## Your assignment (planning only — DO NOT implement)",
|
|
829
|
+
"",
|
|
830
|
+
`Assigned issues: ${assigned}. The real title/body of each is inline below.`,
|
|
831
|
+
"",
|
|
832
|
+
...(issueBlocks.length > 0
|
|
833
|
+
? issueBlocks
|
|
834
|
+
: ["(no issue numbers recorded — read ORCHESTRATOR_TASK.md for the full task)"]),
|
|
835
|
+
"",
|
|
836
|
+
"1. Read `ORCHESTRATOR_TASK.md` at the workspace root — it is the FULL task the code session",
|
|
837
|
+
" must execute. Read the assigned issues' inline bodies above.",
|
|
838
|
+
"2. Explore the codebase just enough to ground the plan (grep/read the files the issues touch).",
|
|
839
|
+
"3. Write the plan to `PLAN.md` at the workspace root (OVERWRITE any existing PLAN.md) as a",
|
|
840
|
+
" compact markdown document: goal, concrete steps (files + functions), verification steps",
|
|
841
|
+
" (tests/boot), and any risks. A code session will execute it — make every step something",
|
|
842
|
+
" another session can act on without re-doing your exploration.",
|
|
843
|
+
"4. Finish with attempt_completion summarizing where the plan was written.",
|
|
844
|
+
"",
|
|
845
|
+
"## Scratch",
|
|
846
|
+
"",
|
|
847
|
+
"Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
|
|
848
|
+
"`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
|
|
849
|
+
"worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
|
|
850
|
+
"",
|
|
851
|
+
"## Constraints",
|
|
852
|
+
"",
|
|
853
|
+
"- DO NOT implement anything: no source edits, no commits, no pushes, no PRs.",
|
|
854
|
+
"- DO NOT ask the human questions — you are headless; decide autonomously and write the plan.",
|
|
855
|
+
"- DO NOT use switch_mode — this session ends with attempt_completion, period.",
|
|
856
|
+
"- Keep the plan SHORT (a compact plan beats a tome; the code session re-verifies anything it",
|
|
857
|
+
" doubts before trusting it).",
|
|
858
|
+
"",
|
|
859
|
+
`Scope: ${assigned} only — planning phase.`,
|
|
860
|
+
].join("\n")
|
|
861
|
+
}
|
|
862
|
+
|
|
863
|
+
export function writeTaskFiles(
|
|
864
|
+
repo: string,
|
|
865
|
+
specs: WorktreeSpec[],
|
|
866
|
+
issues: SplitIssue[],
|
|
867
|
+
opts: { planFirst?: boolean } = {},
|
|
868
|
+
): string[] {
|
|
869
|
+
const dir = path.join(repo, "plans", "parallel-tasks")
|
|
870
|
+
fs.mkdirSync(dir, { recursive: true })
|
|
871
|
+
const written: string[] = []
|
|
872
|
+
for (const spec of specs) {
|
|
873
|
+
const target = path.join(dir, spec.taskFile)
|
|
874
|
+
fs.writeFileSync(target, buildTaskFileContent(spec, issues), "utf-8")
|
|
875
|
+
written.push(target)
|
|
876
|
+
if (opts.planFirst) {
|
|
877
|
+
// Issue #49: the plan-first session's task file (the spawner copies
|
|
878
|
+
// it into the worktree as ORCHESTRATOR_PLAN.md and runs it BEFORE
|
|
879
|
+
// the code worker). Named `<name>-plan.md` — the spawner derives it
|
|
880
|
+
// from the worktree name alone, so a standalone spawner invocation
|
|
881
|
+
// can fall back to a generic planner prompt when it is absent.
|
|
882
|
+
const planTarget = path.join(dir, `${spec.name}-plan.md`)
|
|
883
|
+
fs.writeFileSync(planTarget, buildPlanFirstTaskFileContent(spec, issues), "utf-8")
|
|
884
|
+
written.push(planTarget)
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
return written
|
|
888
|
+
}
|
|
889
|
+
|
|
890
|
+
/**
|
|
891
|
+
* Rework-loop task content (see plans/rework-loop.md): a NEW task file for a
|
|
892
|
+
* group whose review came back with a "finding" verdict. Reuses the original
|
|
893
|
+
* task file's structure/conventions (rules preamble, closing-report template,
|
|
894
|
+
* scope) but the assignment is the reviewer's `pending_review_findings` list
|
|
895
|
+
* rather than the raw issues. The worker is expected to address EVERY finding,
|
|
896
|
+
* re-verify for real, and re-close/re-comment with fresh evidence per the
|
|
897
|
+
* closing-report template — same contract as the first attempt.
|
|
898
|
+
*/
|
|
899
|
+
export function buildReworkTaskFileContent(
|
|
900
|
+
group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
|
|
901
|
+
findings: string[],
|
|
902
|
+
reworkCount: number,
|
|
903
|
+
/** Persisted issue title/body captured at dispatch time (see
|
|
904
|
+
* OrchestratorGroup.issueBodies). Absent for state written before the
|
|
905
|
+
* bodies feature — falls back to the old `gh issue view <n>` instruction. */
|
|
906
|
+
issueBodies?: IssueBodies,
|
|
907
|
+
): string {
|
|
908
|
+
const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
|
|
909
|
+
const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
|
|
910
|
+
const findingsList = findings.length > 0 ? findings.map((f, i) => `${i + 1}. ${f}`).join("\n") : ""
|
|
911
|
+
return [
|
|
912
|
+
...RULES_SPLICED_NOTE,
|
|
913
|
+
"",
|
|
914
|
+
"You are working autonomously in the SAME isolated git worktree as the previous attempt, on",
|
|
915
|
+
"the same branch. The automated adversarial reviewer found problems with that attempt. Read",
|
|
916
|
+
"this entire file before doing anything.",
|
|
917
|
+
"",
|
|
918
|
+
`## Rework cycle ${reworkCount} — address the review findings`,
|
|
919
|
+
"",
|
|
920
|
+
`Assigned issues: ${assigned}. The real title/body of each is inline below.`,
|
|
921
|
+
"",
|
|
922
|
+
...(issueBlocks.length > 0
|
|
923
|
+
? issueBlocks
|
|
924
|
+
: ["(no issue numbers recorded — inspect the worktree for context)"]),
|
|
925
|
+
"",
|
|
926
|
+
"## Review findings to fix (from the automated review)",
|
|
927
|
+
"",
|
|
928
|
+
...findingsList.split("\n"),
|
|
929
|
+
...(findingsList === "" ? ["(no specific findings were recorded — re-read the previous attempt's", "diff and re-verify everything the reviewer might have flagged.)"] : []),
|
|
930
|
+
"",
|
|
931
|
+
"Your job: address EVERY finding above. Do not skip any. Do not claim a finding is fixed",
|
|
932
|
+
"without reproducing the fix for real.",
|
|
933
|
+
"",
|
|
934
|
+
"## Scratch",
|
|
935
|
+
"",
|
|
936
|
+
"Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
|
|
937
|
+
"`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
|
|
938
|
+
"worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
|
|
939
|
+
"",
|
|
940
|
+
"## Workflow",
|
|
941
|
+
"",
|
|
942
|
+
"1. Read each finding above and the relevant source files.",
|
|
943
|
+
"2. Fix every finding, following the project's own conventions and the rules files.",
|
|
944
|
+
"3. Re-verify for real (run the project's tests / boot check) and capture the output —",
|
|
945
|
+
" output in a report is a claim, not evidence, until reproduced.",
|
|
946
|
+
"4. Commit with the issue number referenced, one logical change per commit.",
|
|
947
|
+
"5. Push the branch and update the PR (or open a new one if needed).",
|
|
948
|
+
"6. Post an UPDATED closing comment below on each issue using the required template, then",
|
|
949
|
+
" re-close it (or leave it open with the updated report if it must stay open).",
|
|
950
|
+
"",
|
|
951
|
+
"## Required comment template",
|
|
952
|
+
"",
|
|
953
|
+
"Your issue-closing comment must use this exact structure (every section present; a section can",
|
|
954
|
+
"be 'N/A' with a reason, but a missing section is itself a review finding):",
|
|
955
|
+
"```",
|
|
956
|
+
`## Issue <n> — Closing Report (rework cycle ${reworkCount})`,
|
|
957
|
+
"",
|
|
958
|
+
"### Files changed",
|
|
959
|
+
"<one line per file>",
|
|
960
|
+
"",
|
|
961
|
+
"### Baseline (before)",
|
|
962
|
+
"<real output>",
|
|
963
|
+
"",
|
|
964
|
+
"### Baseline (after)",
|
|
965
|
+
"<real output>",
|
|
966
|
+
"",
|
|
967
|
+
"### Boot check",
|
|
968
|
+
"<real output or exact reason it didn't apply>",
|
|
969
|
+
"",
|
|
970
|
+
"### Residual risk / notes",
|
|
971
|
+
"<anything not 100% certain>",
|
|
972
|
+
"",
|
|
973
|
+
"PR: <link>",
|
|
974
|
+
"```",
|
|
975
|
+
"",
|
|
976
|
+
"## Closing out",
|
|
977
|
+
"",
|
|
978
|
+
"Once every finding is addressed and every assigned issue is re-closed with fresh evidence:",
|
|
979
|
+
"1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
|
|
980
|
+
" numbers, the boot check, and the PR link(s).",
|
|
981
|
+
"2. The harness records your completion automatically (exit code + .harness.done marker) —",
|
|
982
|
+
" no separate phone-home step is needed.",
|
|
983
|
+
"3. Do not start on any issue not assigned to this worktree.",
|
|
984
|
+
"",
|
|
985
|
+
`Scope: ${assigned} only — rework cycle ${reworkCount}.`,
|
|
986
|
+
].join("\n")
|
|
987
|
+
}
|
|
988
|
+
|
|
989
|
+
/**
|
|
990
|
+
* QA-fail rework task content (issue #52): a NEW task file for a group whose
|
|
991
|
+
* QA session came back with a real "fail" verdict (review had passed). Same
|
|
992
|
+
* structure/conventions as buildReworkTaskFileContent, but the assignment is
|
|
993
|
+
* the QA evidence — what the QA session actually found and verified — rather
|
|
994
|
+
* than the reviewer's findings list. The worker is expected to address every
|
|
995
|
+
* piece of evidence, re-verify for real, and re-close with fresh evidence per
|
|
996
|
+
* the closing-report template — same contract as the first attempt.
|
|
997
|
+
*/
|
|
998
|
+
export function buildQaReworkTaskFileContent(
|
|
999
|
+
group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
|
|
1000
|
+
evidence: string,
|
|
1001
|
+
reworkCount: number,
|
|
1002
|
+
reportPath?: string,
|
|
1003
|
+
/** Persisted issue title/body captured at dispatch time (see
|
|
1004
|
+
* OrchestratorGroup.issueBodies). Absent for state written before the
|
|
1005
|
+
* bodies feature — falls back to the old `gh issue view <n>` instruction. */
|
|
1006
|
+
issueBodies?: IssueBodies,
|
|
1007
|
+
): string {
|
|
1008
|
+
const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
|
|
1009
|
+
const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
|
|
1010
|
+
// Issue #34-style: point at the QA session's COMPLETE final report so the
|
|
1011
|
+
// full reasoning is one file-read away (evidence is only a 4000-char slice).
|
|
1012
|
+
const reportLines = reportPath
|
|
1013
|
+
? [
|
|
1014
|
+
"",
|
|
1015
|
+
`The full QA session report is at: \`${reportPath}\` — read it for the complete`,
|
|
1016
|
+
"reasoning behind every finding before you start.",
|
|
1017
|
+
]
|
|
1018
|
+
: []
|
|
1019
|
+
return [
|
|
1020
|
+
...RULES_SPLICED_NOTE,
|
|
1021
|
+
"",
|
|
1022
|
+
"You are working autonomously in the SAME isolated git worktree as the previous attempt, on",
|
|
1023
|
+
"the same branch. Review passed, but the automated QA session found real problems with the",
|
|
1024
|
+
"attempt — treat them as NEW WORK. Read this entire file before doing anything.",
|
|
1025
|
+
"",
|
|
1026
|
+
`## Rework cycle ${reworkCount} — address the QA findings`,
|
|
1027
|
+
"",
|
|
1028
|
+
`Assigned issues: ${assigned}. The real title/body of each is inline below.`,
|
|
1029
|
+
"",
|
|
1030
|
+
...(issueBlocks.length > 0
|
|
1031
|
+
? issueBlocks
|
|
1032
|
+
: ["(no issue numbers recorded — inspect the worktree for context)"]),
|
|
1033
|
+
"",
|
|
1034
|
+
"## QA findings to fix (from the automated QA)",
|
|
1035
|
+
"",
|
|
1036
|
+
...(evidence === ""
|
|
1037
|
+
? [
|
|
1038
|
+
"(the QA report recorded no extractable evidence — re-run the QA checklist against the",
|
|
1039
|
+
"worktree yourself and fix whatever fails.)",
|
|
1040
|
+
]
|
|
1041
|
+
: [evidence]),
|
|
1042
|
+
...reportLines,
|
|
1043
|
+
"",
|
|
1044
|
+
"Your job: address EVERY finding above. Do not skip any. Do not claim a finding is fixed",
|
|
1045
|
+
"without reproducing the fix for real.",
|
|
1046
|
+
"",
|
|
1047
|
+
"## Scratch",
|
|
1048
|
+
"",
|
|
1049
|
+
"Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
|
|
1050
|
+
"`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
|
|
1051
|
+
"worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
|
|
1052
|
+
"",
|
|
1053
|
+
"## Workflow",
|
|
1054
|
+
"",
|
|
1055
|
+
"1. Read each finding above and the relevant source files.",
|
|
1056
|
+
"2. Fix every finding, following the project's own conventions and the rules files.",
|
|
1057
|
+
"3. Re-verify for real (run the project's tests / boot check) and capture the output —",
|
|
1058
|
+
" output in a report is a claim, not evidence, until reproduced.",
|
|
1059
|
+
"4. Commit with the issue number referenced, one logical change per commit.",
|
|
1060
|
+
"5. Push the branch and update the PR (or open a new one if needed).",
|
|
1061
|
+
"6. Post an UPDATED closing comment below on each issue using the required template, then",
|
|
1062
|
+
" re-close it (or leave it open with the updated report if it must stay open).",
|
|
1063
|
+
"",
|
|
1064
|
+
"## Required comment template",
|
|
1065
|
+
"",
|
|
1066
|
+
"Your issue-closing comment must use this exact structure (every section present; a section can",
|
|
1067
|
+
"be 'N/A' with a reason, but a missing section is itself a review finding):",
|
|
1068
|
+
"```",
|
|
1069
|
+
`## Issue <n> — Closing Report (rework cycle ${reworkCount})`,
|
|
1070
|
+
"",
|
|
1071
|
+
"### Files changed",
|
|
1072
|
+
"<one line per file>",
|
|
1073
|
+
"",
|
|
1074
|
+
"### Baseline (before)",
|
|
1075
|
+
"<real output>",
|
|
1076
|
+
"",
|
|
1077
|
+
"### Baseline (after)",
|
|
1078
|
+
"<real output>",
|
|
1079
|
+
"",
|
|
1080
|
+
"### Boot check",
|
|
1081
|
+
"<real output or exact reason it didn't apply>",
|
|
1082
|
+
"",
|
|
1083
|
+
"### Residual risk / notes",
|
|
1084
|
+
"<anything not 100% certain>",
|
|
1085
|
+
"",
|
|
1086
|
+
"PR: <link>",
|
|
1087
|
+
"```",
|
|
1088
|
+
"",
|
|
1089
|
+
"## Closing out",
|
|
1090
|
+
"",
|
|
1091
|
+
"Once every finding is addressed and every assigned issue is re-closed with fresh evidence:",
|
|
1092
|
+
"1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
|
|
1093
|
+
" numbers, the boot check, and the PR link(s).",
|
|
1094
|
+
"2. The harness records your completion automatically (exit code + .harness.done marker) —",
|
|
1095
|
+
" no separate phone-home step is needed.",
|
|
1096
|
+
"3. Do not start on any issue not assigned to this worktree.",
|
|
1097
|
+
"",
|
|
1098
|
+
`Scope: ${assigned} only — rework cycle ${reworkCount}.`,
|
|
1099
|
+
].join("\n")
|
|
1100
|
+
}
|
|
1101
|
+
|
|
1102
|
+
/**
|
|
1103
|
+
* Continuation task content (plans/issues/02-orchestrate-iteration-plumbing.md,
|
|
1104
|
+
* Part 2): a NEW task file for a group whose worker exited because it hit the
|
|
1105
|
+
* iteration cap. The worktree is untouched and resumable — the partial edits
|
|
1106
|
+
* are on disk — so the next session picks up where the last left off. Reuses
|
|
1107
|
+
* the original task file's conventions (rules preamble, closing-report
|
|
1108
|
+
* template, scope) but the assignment is "keep going" rather than the raw
|
|
1109
|
+
* issues.
|
|
1110
|
+
*/
|
|
1111
|
+
export function buildContinuationTaskFileContent(
|
|
1112
|
+
group: { name: string; issues?: number[]; issueBodies?: IssueBodies },
|
|
1113
|
+
continuationCount: number,
|
|
1114
|
+
/** Persisted issue title/body captured at dispatch time (see
|
|
1115
|
+
* OrchestratorGroup.issueBodies). Absent for state written before the
|
|
1116
|
+
* bodies feature — falls back to the old `gh issue view <n>` instruction. */
|
|
1117
|
+
issueBodies?: IssueBodies,
|
|
1118
|
+
/** The previous session's own condensed summary of its history (see
|
|
1119
|
+
* src/engine/handoff.ts) — file reads, command outputs, decisions made —
|
|
1120
|
+
* so this cycle doesn't have to re-derive them from scratch. Absent when
|
|
1121
|
+
* the previous session's handoff write failed/didn't run (older state,
|
|
1122
|
+
* or a very short session); the git-status-inspection instruction below
|
|
1123
|
+
* is the fallback either way, not a redundant step. */
|
|
1124
|
+
handoffSummary?: string,
|
|
1125
|
+
): string {
|
|
1126
|
+
const assigned = (group.issues ?? []).map((n) => `#${n}`).join(", ") || "this worktree"
|
|
1127
|
+
const issueBlocks = issueBodyBlocks(group.issues ?? [], (n) => group.issueBodies?.[String(n)] ?? issueBodies?.[String(n)])
|
|
1128
|
+
return [
|
|
1129
|
+
...RULES_SPLICED_NOTE,
|
|
1130
|
+
"",
|
|
1131
|
+
"You are working autonomously in the SAME isolated git worktree as the previous session(s), on",
|
|
1132
|
+
"the same branch. The previous session was cut short — either it hit the harness's iteration cap, or",
|
|
1133
|
+
"it hit a transient LLM-provider failure (a hard-pinned model with no fallback provider had a bad",
|
|
1134
|
+
"moment; not a fault in your task or the code). Neither is a review verdict — its partial work is",
|
|
1135
|
+
"still on disk: uncommitted edits, commits, notes — pick up exactly where it left off. Read this",
|
|
1136
|
+
"entire file before doing anything.",
|
|
1137
|
+
"",
|
|
1138
|
+
`## Continuation cycle ${continuationCount} — complete the task`,
|
|
1139
|
+
"",
|
|
1140
|
+
`Assigned issues: ${assigned}. The real title/body of each is inline below.`,
|
|
1141
|
+
"",
|
|
1142
|
+
...(issueBlocks.length > 0
|
|
1143
|
+
? issueBlocks
|
|
1144
|
+
: ["(no issue numbers recorded — inspect the worktree for context)"]),
|
|
1145
|
+
"",
|
|
1146
|
+
"## What the previous session already learned — do not re-derive this",
|
|
1147
|
+
"",
|
|
1148
|
+
...(handoffSummary
|
|
1149
|
+
? [
|
|
1150
|
+
"The previous session summarized its own history before it ran out of iterations. Trust it and",
|
|
1151
|
+
"build on it — re-reading files or re-running commands it already covered wastes iterations you",
|
|
1152
|
+
"need for finishing the actual task. It CAN be wrong (compression loses detail, and the session's",
|
|
1153
|
+
"own notes may be incomplete) — if something below conflicts with what you observe on disk, what",
|
|
1154
|
+
"you observe wins, but don't re-verify things it reports with confidence just to be safe.",
|
|
1155
|
+
"",
|
|
1156
|
+
"```",
|
|
1157
|
+
handoffSummary.trim(),
|
|
1158
|
+
"```",
|
|
1159
|
+
]
|
|
1160
|
+
: [
|
|
1161
|
+
"(No handoff summary available for this cycle — the previous session's write either failed or",
|
|
1162
|
+
"never ran. Fall back to inspecting the worktree yourself, below.)",
|
|
1163
|
+
]),
|
|
1164
|
+
"",
|
|
1165
|
+
"## What the previous session did",
|
|
1166
|
+
"",
|
|
1167
|
+
"- Inspect the worktree's git status, recent commits, and uncommitted changes to see exactly",
|
|
1168
|
+
" how far the previous session got — this is a real-state CHECK, not your primary source of",
|
|
1169
|
+
" context; the summary above should already tell you most of this.",
|
|
1170
|
+
"- Do NOT restart from scratch: continue the existing work toward the issues above.",
|
|
1171
|
+
"- If the previous session left a partial fix, finish it and verify it for real.",
|
|
1172
|
+
"",
|
|
1173
|
+
"## Scratch",
|
|
1174
|
+
"",
|
|
1175
|
+
"Any scratch/temporary files (probe scripts, one-off data dumps, intermediate output) go in",
|
|
1176
|
+
"`<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside this",
|
|
1177
|
+
"worktree. `/.headlesscode/` is gitignored, so scratch there needs no ignore rule.",
|
|
1178
|
+
"",
|
|
1179
|
+
"## Workflow",
|
|
1180
|
+
"",
|
|
1181
|
+
"1. Read each assigned issue's inline body above and the relevant source files.",
|
|
1182
|
+
"2. Continue the implementation following the project's own conventions and the rules files.",
|
|
1183
|
+
"3. Verify for real (run the project's tests / boot check) and capture the output —",
|
|
1184
|
+
" output in a report is a claim, not evidence, until reproduced.",
|
|
1185
|
+
"4. Commit with the issue number referenced, one logical change per commit.",
|
|
1186
|
+
"5. Push the branch and open a PR (ready, not draft).",
|
|
1187
|
+
"6. Post the closing comment below on each issue using the required template, then close it.",
|
|
1188
|
+
"",
|
|
1189
|
+
"## Required comment template",
|
|
1190
|
+
"",
|
|
1191
|
+
"Your issue-closing comment must use this exact structure (every section present; a section can",
|
|
1192
|
+
"be 'N/A' with a reason, but a missing section is itself a review finding):",
|
|
1193
|
+
"```",
|
|
1194
|
+
`## Issue <n> — Closing Report (continuation cycle ${continuationCount})`,
|
|
1195
|
+
"",
|
|
1196
|
+
"### Files changed",
|
|
1197
|
+
"<one line per file>",
|
|
1198
|
+
"",
|
|
1199
|
+
"### Baseline (before)",
|
|
1200
|
+
"<real output>",
|
|
1201
|
+
"",
|
|
1202
|
+
"### Baseline (after)",
|
|
1203
|
+
"<real output>",
|
|
1204
|
+
"",
|
|
1205
|
+
"### Boot check",
|
|
1206
|
+
"<real output or exact reason it didn't apply>",
|
|
1207
|
+
"",
|
|
1208
|
+
"### Residual risk / notes",
|
|
1209
|
+
"<anything not 100% certain>",
|
|
1210
|
+
"",
|
|
1211
|
+
"PR: <link>",
|
|
1212
|
+
"```",
|
|
1213
|
+
"",
|
|
1214
|
+
"## Closing out",
|
|
1215
|
+
"",
|
|
1216
|
+
"Once every assigned issue is closed with the template above:",
|
|
1217
|
+
"1. Finish with attempt_completion and a full summary of what you changed, the real baseline",
|
|
1218
|
+
" numbers, the boot check, and the PR link(s).",
|
|
1219
|
+
"2. The harness records your completion automatically (exit code + .harness.done marker) —",
|
|
1220
|
+
" no separate phone-home step is needed.",
|
|
1221
|
+
"3. Do not start on any issue not assigned to this worktree.",
|
|
1222
|
+
"",
|
|
1223
|
+
`Scope: ${assigned} only — continuation cycle ${continuationCount}.`,
|
|
1224
|
+
].join("\n")
|
|
1225
|
+
}
|
|
1226
|
+
|
|
1227
|
+
// ─── Rework-loop decision ────────────────────────────────────────────────────
|
|
1228
|
+
|
|
1229
|
+
/**
|
|
1230
|
+
* The outcome of deciding what to do with a group whose review came back with
|
|
1231
|
+
* a "finding" verdict (see plans/rework-loop.md). Either the group gets a NEW
|
|
1232
|
+
* rework attempt on the SAME worktree (shouldSpawn), or it has exhausted its
|
|
1233
|
+
* `--max-rework-cycles` budget and is marked for human attention.
|
|
1234
|
+
*/
|
|
1235
|
+
export interface ReworkDecision {
|
|
1236
|
+
/** State patch to apply to the group (rework reset, or needs-human). */
|
|
1237
|
+
patch: Partial<Omit<OrchestratorGroup, "name">>
|
|
1238
|
+
/** Whether a rework worker should be spawned on the same worktree. */
|
|
1239
|
+
shouldSpawn: boolean
|
|
1240
|
+
/** The group's reworkCount after this decision. */
|
|
1241
|
+
newReworkCount: number
|
|
1242
|
+
/** Absolute path of the rework task file (only when shouldSpawn). */
|
|
1243
|
+
taskFilePath?: string
|
|
1244
|
+
/** Content of the rework task file (only when shouldSpawn). */
|
|
1245
|
+
taskContent?: string
|
|
1246
|
+
/** Exact `bash -c` command that launches the rework worker (only when shouldSpawn). */
|
|
1247
|
+
spawnCommand?: string
|
|
1248
|
+
}
|
|
1249
|
+
|
|
1250
|
+
/**
|
|
1251
|
+
* Decide what to do after a review "finding" verdict, and build the pieces the
|
|
1252
|
+
* caller needs (state patch, task file, spawn command). Extracted from the
|
|
1253
|
+
* watchGroups callback so every branch is unit-testable without a live round.
|
|
1254
|
+
*
|
|
1255
|
+
* - Below the cap: the group is reset to "running" on the SAME worktree with a
|
|
1256
|
+
* fresh `spawned` timestamp (so the watcher's stall guard measures the rework
|
|
1257
|
+
* worker, not the original one), review/QA records are cleared, reworkCount is
|
|
1258
|
+
* incremented, and a rework task file + worker launch command are produced.
|
|
1259
|
+
* - At the cap: no spawn; the group is marked `needs-human` (a TERMINAL state,
|
|
1260
|
+
* deliberately distinct from `failed`) with the findings left recorded so a
|
|
1261
|
+
* human can see exactly what the reviewer flagged.
|
|
1262
|
+
*/
|
|
1263
|
+
export function handleReviewVerdict(
|
|
1264
|
+
group: OrchestratorGroup,
|
|
1265
|
+
result: ReviewResult,
|
|
1266
|
+
repo: string,
|
|
1267
|
+
maxReworkCycles: number,
|
|
1268
|
+
mode: string,
|
|
1269
|
+
model?: string,
|
|
1270
|
+
/** Per-session iteration cap forwarded to the rework worker (default: harness's own). */
|
|
1271
|
+
maxIterations?: number,
|
|
1272
|
+
): ReworkDecision {
|
|
1273
|
+
const currentRework = typeof group.reworkCount === "number" ? group.reworkCount : 0
|
|
1274
|
+
const newReworkCount = currentRework + 1
|
|
1275
|
+
const now = new Date().toISOString()
|
|
1276
|
+
|
|
1277
|
+
if (currentRework < maxReworkCycles) {
|
|
1278
|
+
const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-rework${newReworkCount}.md`)
|
|
1279
|
+
const taskContent = buildReworkTaskFileContent(group, result.findings, newReworkCount, group.issueBodies)
|
|
1280
|
+
const worktreePath = groupWorktreePath(repo, group)
|
|
1281
|
+
const spawnCommand =
|
|
1282
|
+
`bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
|
|
1283
|
+
`"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
|
|
1284
|
+
(model ? ` --model "${model}"` : "") +
|
|
1285
|
+
(maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
|
|
1286
|
+
return {
|
|
1287
|
+
shouldSpawn: true,
|
|
1288
|
+
newReworkCount,
|
|
1289
|
+
taskFilePath,
|
|
1290
|
+
taskContent,
|
|
1291
|
+
spawnCommand,
|
|
1292
|
+
patch: {
|
|
1293
|
+
status: "running",
|
|
1294
|
+
reworkCount: newReworkCount,
|
|
1295
|
+
// Fresh spawned timestamp: the stall guard must measure the
|
|
1296
|
+
// rework worker, not the (possibly hours-old) original attempt.
|
|
1297
|
+
spawned: now,
|
|
1298
|
+
// A fresh cycle needs a fresh review — clear the old verdict,
|
|
1299
|
+
// findings, QA, and the previous attempt's completion artifacts.
|
|
1300
|
+
review_verdict: undefined,
|
|
1301
|
+
pending_review_findings: undefined,
|
|
1302
|
+
reviewed_at: undefined,
|
|
1303
|
+
review_report: undefined,
|
|
1304
|
+
qa: undefined,
|
|
1305
|
+
stalled: undefined,
|
|
1306
|
+
exit_code: undefined,
|
|
1307
|
+
summary: undefined,
|
|
1308
|
+
// Must also clear cost_recorded: the rework worker adds MORE real
|
|
1309
|
+
// cost on this worktree, and recordCostIfSettled's one-shot gate
|
|
1310
|
+
// would otherwise silently skip re-recording the combined total
|
|
1311
|
+
// once this cycle re-settles — a real bug caught while manually
|
|
1312
|
+
// recovering a stuck round (see cost-history.ts).
|
|
1313
|
+
cost_recorded: undefined,
|
|
1314
|
+
last_activity: {
|
|
1315
|
+
note: `rework cycle ${newReworkCount}: re-spawned worker on the same worktree to fix review findings`,
|
|
1316
|
+
},
|
|
1317
|
+
},
|
|
1318
|
+
}
|
|
1319
|
+
}
|
|
1320
|
+
|
|
1321
|
+
// Cap reached: no new attempt. Terminal "needs-human" outcome — the
|
|
1322
|
+
// findings stay recorded so a human can see exactly what failed.
|
|
1323
|
+
return {
|
|
1324
|
+
shouldSpawn: false,
|
|
1325
|
+
newReworkCount: currentRework,
|
|
1326
|
+
patch: {
|
|
1327
|
+
status: "needs-human",
|
|
1328
|
+
reworkCount: currentRework,
|
|
1329
|
+
last_activity: {
|
|
1330
|
+
note: `exhausted ${currentRework} rework attempt(s); review still has ${result.findings.length} finding(s) — needs human`,
|
|
1331
|
+
},
|
|
1332
|
+
},
|
|
1333
|
+
}
|
|
1334
|
+
}
|
|
1335
|
+
|
|
1336
|
+
/**
|
|
1337
|
+
* QA-side twin of handleReviewVerdict (issue #52): a REAL QA "fail" verdict
|
|
1338
|
+
* (as opposed to a QA-SESSION "error" — see handleQaSessionError) is
|
|
1339
|
+
* actionable new work, exactly like a review "finding". Below the
|
|
1340
|
+
* `--max-rework-cycles` cap the group is reset to "running" on the SAME
|
|
1341
|
+
* worktree with a fresh `spawned` timestamp, review/QA/cost artifacts are
|
|
1342
|
+
* cleared, reworkCount is incremented, and a QA rework task file (fed from
|
|
1343
|
+
* `qa.evidence` rather than review findings) + worker launch command are
|
|
1344
|
+
* produced. At the cap: no spawn; the group is marked `needs-human` with the
|
|
1345
|
+
* QA verdict/evidence left recorded so a human can see exactly what failed.
|
|
1346
|
+
*/
|
|
1347
|
+
export function handleQaVerdict(
|
|
1348
|
+
group: OrchestratorGroup,
|
|
1349
|
+
result: QaResult,
|
|
1350
|
+
repo: string,
|
|
1351
|
+
maxReworkCycles: number,
|
|
1352
|
+
mode: string,
|
|
1353
|
+
model?: string,
|
|
1354
|
+
/** Per-session iteration cap forwarded to the rework worker (default: harness's own). */
|
|
1355
|
+
maxIterations?: number,
|
|
1356
|
+
): ReworkDecision {
|
|
1357
|
+
const currentRework = typeof group.reworkCount === "number" ? group.reworkCount : 0
|
|
1358
|
+
const newReworkCount = currentRework + 1
|
|
1359
|
+
const now = new Date().toISOString()
|
|
1360
|
+
|
|
1361
|
+
if (currentRework < maxReworkCycles) {
|
|
1362
|
+
const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-qa-rework${newReworkCount}.md`)
|
|
1363
|
+
const taskContent = buildQaReworkTaskFileContent(
|
|
1364
|
+
group,
|
|
1365
|
+
result.evidence,
|
|
1366
|
+
newReworkCount,
|
|
1367
|
+
result.reportPath,
|
|
1368
|
+
group.issueBodies,
|
|
1369
|
+
)
|
|
1370
|
+
const worktreePath = groupWorktreePath(repo, group)
|
|
1371
|
+
const spawnCommand =
|
|
1372
|
+
`bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
|
|
1373
|
+
`"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
|
|
1374
|
+
(model ? ` --model "${model}"` : "") +
|
|
1375
|
+
(maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
|
|
1376
|
+
return {
|
|
1377
|
+
shouldSpawn: true,
|
|
1378
|
+
newReworkCount,
|
|
1379
|
+
taskFilePath,
|
|
1380
|
+
taskContent,
|
|
1381
|
+
spawnCommand,
|
|
1382
|
+
patch: {
|
|
1383
|
+
status: "running",
|
|
1384
|
+
reworkCount: newReworkCount,
|
|
1385
|
+
// Fresh spawned timestamp: the stall guard must measure the
|
|
1386
|
+
// rework worker, not the (possibly hours-old) original attempt.
|
|
1387
|
+
spawned: now,
|
|
1388
|
+
// The rework worker changes code, so review AND QA must both
|
|
1389
|
+
// re-run — same fresh-cycle reset as handleReviewVerdict.
|
|
1390
|
+
review_verdict: undefined,
|
|
1391
|
+
pending_review_findings: undefined,
|
|
1392
|
+
reviewed_at: undefined,
|
|
1393
|
+
review_report: undefined,
|
|
1394
|
+
qa: undefined,
|
|
1395
|
+
stalled: undefined,
|
|
1396
|
+
exit_code: undefined,
|
|
1397
|
+
summary: undefined,
|
|
1398
|
+
// The rework worker adds MORE real cost on this worktree; the
|
|
1399
|
+
// one-shot recordCostIfSettled gate must re-record once this
|
|
1400
|
+
// cycle re-settles (same reasoning as handleReviewVerdict).
|
|
1401
|
+
cost_recorded: undefined,
|
|
1402
|
+
last_activity: {
|
|
1403
|
+
note: `rework cycle ${newReworkCount}: re-spawned worker on the same worktree to fix QA findings`,
|
|
1404
|
+
},
|
|
1405
|
+
},
|
|
1406
|
+
}
|
|
1407
|
+
}
|
|
1408
|
+
|
|
1409
|
+
// Cap reached: no new attempt. Terminal "needs-human" outcome — the QA
|
|
1410
|
+
// verdict/evidence stay recorded so a human can see exactly what failed.
|
|
1411
|
+
return {
|
|
1412
|
+
shouldSpawn: false,
|
|
1413
|
+
newReworkCount: currentRework,
|
|
1414
|
+
patch: {
|
|
1415
|
+
status: "needs-human",
|
|
1416
|
+
reworkCount: currentRework,
|
|
1417
|
+
qa: {
|
|
1418
|
+
status: "failed",
|
|
1419
|
+
verdict: result.verdict,
|
|
1420
|
+
evidence: result.evidence.slice(0, 4000),
|
|
1421
|
+
report: result.reportPath,
|
|
1422
|
+
updated: now,
|
|
1423
|
+
},
|
|
1424
|
+
last_activity: {
|
|
1425
|
+
note: `exhausted ${currentRework} rework attempt(s); QA still failing — needs human`,
|
|
1426
|
+
},
|
|
1427
|
+
},
|
|
1428
|
+
}
|
|
1429
|
+
}
|
|
1430
|
+
|
|
1431
|
+
/**
|
|
1432
|
+
* A review-SESSION failure (crash/budget/mistake-limit — see
|
|
1433
|
+
* runReviewWithRetries) survived every retry: verdict "error", distinct from
|
|
1434
|
+
* a real code "finding". This must NEVER feed into handleReviewVerdict's
|
|
1435
|
+
* rework-a-worker path — there is nothing actionable in a placeholder error
|
|
1436
|
+
* message, and spawning a worker to "fix" it wastes a full session for
|
|
1437
|
+
* nothing (caught live 2026-08-05 on issue #17's own round: a review
|
|
1438
|
+
* session's own bounded-failure triggered a pointless worker rework cycle).
|
|
1439
|
+
* Always terminal needs-human — a human should look at why review sessions
|
|
1440
|
+
* keep failing against this worktree, not have a worker respawned blindly.
|
|
1441
|
+
*/
|
|
1442
|
+
export function handleReviewSessionError(result: ReviewResult): Partial<OrchestratorGroup> {
|
|
1443
|
+
return {
|
|
1444
|
+
status: "needs-human",
|
|
1445
|
+
review_verdict: result.verdict,
|
|
1446
|
+
pending_review_findings: result.findings,
|
|
1447
|
+
reviewed_at: new Date().toISOString(),
|
|
1448
|
+
last_activity: { note: `review session kept failing: ${result.summary}` },
|
|
1449
|
+
}
|
|
1450
|
+
}
|
|
1451
|
+
|
|
1452
|
+
/**
|
|
1453
|
+
* QA-side twin of handleReviewSessionError: a QA-SESSION failure (verdict
|
|
1454
|
+
* "error" — see runQaWithRetries) survived every retry. Must escalate to
|
|
1455
|
+
* needs-human directly, never recorded as an ordinary "failed" QA (which
|
|
1456
|
+
* would settle the round with cost recorded and nothing left to retry,
|
|
1457
|
+
* while QA never actually validated anything). Caught live 2026-08-05 on
|
|
1458
|
+
* issue #18's own round.
|
|
1459
|
+
*/
|
|
1460
|
+
export function handleQaSessionError(result: QaResult): Partial<OrchestratorGroup> {
|
|
1461
|
+
return {
|
|
1462
|
+
status: "needs-human",
|
|
1463
|
+
qa: {
|
|
1464
|
+
status: "failed",
|
|
1465
|
+
verdict: result.verdict,
|
|
1466
|
+
evidence: result.evidence.slice(0, 4000),
|
|
1467
|
+
updated: new Date().toISOString(),
|
|
1468
|
+
},
|
|
1469
|
+
last_activity: { note: `QA session kept failing: ${result.summary}` },
|
|
1470
|
+
}
|
|
1471
|
+
}
|
|
1472
|
+
|
|
1473
|
+
// ─── Iteration-exhaustion continuation decision ──────────────────────────────
|
|
1474
|
+
|
|
1475
|
+
/**
|
|
1476
|
+
* The outcome of deciding what to do with a group whose worker failed with
|
|
1477
|
+
* exit != 0 (plans/issues/02-orchestrate-iteration-plumbing.md, Part 2).
|
|
1478
|
+
* Either the group gets a NEW continuation session on the SAME worktree
|
|
1479
|
+
* (shouldSpawn), or it has exhausted its `--max-continuations` budget and is
|
|
1480
|
+
* marked for human attention, or the failure is not the iteration-cap reason
|
|
1481
|
+
* at all and the group stays `failed`.
|
|
1482
|
+
*/
|
|
1483
|
+
export interface ContinuationDecision {
|
|
1484
|
+
/** State patch to apply to the group (running reset, needs-human, or empty for a real failure). */
|
|
1485
|
+
patch: Partial<Omit<OrchestratorGroup, "name">>
|
|
1486
|
+
/** Whether a continuation worker should be spawned on the same worktree. */
|
|
1487
|
+
shouldSpawn: boolean
|
|
1488
|
+
/** The group's continuationCount after this decision. */
|
|
1489
|
+
newContinuationCount: number
|
|
1490
|
+
/** Absolute path of the continuation task file (only when shouldSpawn). */
|
|
1491
|
+
taskFilePath?: string
|
|
1492
|
+
/** Content of the continuation task file (only when shouldSpawn). */
|
|
1493
|
+
taskContent?: string
|
|
1494
|
+
/** Exact `bash -c` command that launches the continuation worker (only when shouldSpawn). */
|
|
1495
|
+
spawnCommand?: string
|
|
1496
|
+
}
|
|
1497
|
+
|
|
1498
|
+
/**
|
|
1499
|
+
* Decide what to do after a worker failed with exit != 0, and build the pieces
|
|
1500
|
+
* the caller needs (state patch, task file, spawn command). Extracted from the
|
|
1501
|
+
* watchGroups callback so every branch is unit-testable without a live round.
|
|
1502
|
+
*
|
|
1503
|
+
* - The failure is NEITHER the iteration-cap reason NOR a transient
|
|
1504
|
+
* LLM-provider failure (budget stop, tool-error storm, a real code/task
|
|
1505
|
+
* bug — or a clean exit with either string merely in the log tail): no
|
|
1506
|
+
* continuation; the patch is empty and the watcher's "failed" stands.
|
|
1507
|
+
* Provider failures earned the same auto-continue treatment as the
|
|
1508
|
+
* iteration cap after a live incident (2026-08-08): a hard-pinned model
|
|
1509
|
+
* with no fallback provider hit an HTTP 520 mid-session, the session
|
|
1510
|
+
* died outright (this predates src/engine/loop.ts's callMainLlm retry —
|
|
1511
|
+
* even with that retry, a longer outage still exhausts it), and the
|
|
1512
|
+
* group sat in the generic terminal "failed" state — indistinguishable
|
|
1513
|
+
* from a real bug — until a human happened to read harness.log. Bounded
|
|
1514
|
+
* the same way as the iteration cap (maxContinuations), so a genuinely
|
|
1515
|
+
* dead provider still surfaces to a human eventually, it just gets a
|
|
1516
|
+
* few free retries first instead of zero.
|
|
1517
|
+
* - Below the cap: the group is reset to "running" on the SAME worktree with a
|
|
1518
|
+
* fresh `spawned` timestamp (so the stall guard measures the continuation
|
|
1519
|
+
* worker, not the attempt that hit the cap), continuationCount is
|
|
1520
|
+
* incremented, completion artifacts are cleared, and a continuation task
|
|
1521
|
+
* file + worker launch command are produced.
|
|
1522
|
+
* - At the cap: no spawn; the group is marked `needs-human` (a TERMINAL state,
|
|
1523
|
+
* deliberately distinct from `failed`).
|
|
1524
|
+
*/
|
|
1525
|
+
export function handleIterationExhaustion(
|
|
1526
|
+
group: OrchestratorGroup,
|
|
1527
|
+
repo: string,
|
|
1528
|
+
maxContinuations: number,
|
|
1529
|
+
mode: string,
|
|
1530
|
+
model?: string,
|
|
1531
|
+
maxIterations?: number,
|
|
1532
|
+
): ContinuationDecision {
|
|
1533
|
+
const current = typeof group.continuationCount === "number" ? group.continuationCount : 0
|
|
1534
|
+
if (group.exit_code === 0 || !(isIterationExhaustion(group.summary) || isProviderFailure(group.summary))) {
|
|
1535
|
+
// Real failure — budget stops and every other exit reason stay
|
|
1536
|
+
// terminal. No patch: the watcher's "failed" stands.
|
|
1537
|
+
return { shouldSpawn: false, newContinuationCount: current, patch: {} }
|
|
1538
|
+
}
|
|
1539
|
+
const newContinuationCount = current + 1
|
|
1540
|
+
const now = new Date().toISOString()
|
|
1541
|
+
|
|
1542
|
+
if (current < maxContinuations) {
|
|
1543
|
+
const taskFilePath = path.join(repo, "plans", "parallel-tasks", `${group.name}-continue${newContinuationCount}.md`)
|
|
1544
|
+
const worktreePath = groupWorktreePath(repo, group)
|
|
1545
|
+
// The session that just hit the cap wrote its own condensed history
|
|
1546
|
+
// here (src/engine/handoff.ts) right before exiting — inline it into
|
|
1547
|
+
// the continuation task file so the new conversation doesn't have to
|
|
1548
|
+
// re-derive everything from scratch, then clear it: it's about to be
|
|
1549
|
+
// baked into the task file text, and a stale copy must never leak
|
|
1550
|
+
// into a LATER, unrelated continuation on this same worktree.
|
|
1551
|
+
const handoffSummary = readHandoffSummary(worktreePath)
|
|
1552
|
+
clearHandoffSummary(worktreePath)
|
|
1553
|
+
const taskContent = buildContinuationTaskFileContent(group, newContinuationCount, group.issueBodies, handoffSummary)
|
|
1554
|
+
const spawnCommand =
|
|
1555
|
+
`bash ${path.join(HARNESS_ROOT_TS, "scripts", "run-worker.sh")} ` +
|
|
1556
|
+
`"${worktreePath}" "${taskFilePath}" --mode "${mode}"` +
|
|
1557
|
+
(model ? ` --model "${model}"` : "") +
|
|
1558
|
+
(maxIterations !== undefined ? ` --max-iterations "${maxIterations}"` : "")
|
|
1559
|
+
return {
|
|
1560
|
+
shouldSpawn: true,
|
|
1561
|
+
newContinuationCount,
|
|
1562
|
+
taskFilePath,
|
|
1563
|
+
taskContent,
|
|
1564
|
+
spawnCommand,
|
|
1565
|
+
patch: {
|
|
1566
|
+
status: "running",
|
|
1567
|
+
continuationCount: newContinuationCount,
|
|
1568
|
+
// Fresh spawned timestamp: the stall guard must measure the
|
|
1569
|
+
// continuation worker, not the (possibly hours-old) attempt
|
|
1570
|
+
// that hit the cap.
|
|
1571
|
+
spawned: now,
|
|
1572
|
+
// A fresh session starts clean: drop the previous attempt's
|
|
1573
|
+
// completion artifacts, review/QA records and stall flag.
|
|
1574
|
+
review_verdict: undefined,
|
|
1575
|
+
pending_review_findings: undefined,
|
|
1576
|
+
reviewed_at: undefined,
|
|
1577
|
+
review_report: undefined,
|
|
1578
|
+
qa: undefined,
|
|
1579
|
+
stalled: undefined,
|
|
1580
|
+
exit_code: undefined,
|
|
1581
|
+
summary: undefined,
|
|
1582
|
+
// Must also clear cost_recorded — see the identical note in
|
|
1583
|
+
// handleReviewVerdict's patch above.
|
|
1584
|
+
cost_recorded: undefined,
|
|
1585
|
+
last_activity: {
|
|
1586
|
+
note: `continuation ${newContinuationCount}: previous session hit the iteration cap; re-spawned worker on the same worktree`,
|
|
1587
|
+
},
|
|
1588
|
+
},
|
|
1589
|
+
}
|
|
1590
|
+
}
|
|
1591
|
+
|
|
1592
|
+
// Cap reached: no new attempt. Terminal "needs-human" outcome.
|
|
1593
|
+
return {
|
|
1594
|
+
shouldSpawn: false,
|
|
1595
|
+
newContinuationCount: current,
|
|
1596
|
+
patch: {
|
|
1597
|
+
status: "needs-human",
|
|
1598
|
+
continuationCount: current,
|
|
1599
|
+
last_activity: {
|
|
1600
|
+
note: `exhausted ${current} continuation attempt(s) — still hitting the iteration cap; needs human`,
|
|
1601
|
+
},
|
|
1602
|
+
},
|
|
1603
|
+
}
|
|
1604
|
+
}
|
|
1605
|
+
|
|
1606
|
+
// ─── Spawn command assembly ──────────────────────────────────────────────────
|
|
1607
|
+
|
|
1608
|
+
function spawnCommandFor(harnessRoot: string, specs: WorktreeSpec[]): string {
|
|
1609
|
+
const triples = specs.map((spec) => `${spec.name}:${specs.indexOf(spec)}:plans/parallel-tasks/${spec.taskFile}`)
|
|
1610
|
+
return `bash ${path.join(harnessRoot, "scripts", "spawn-parallel-worktrees.sh")} ${triples.join(" ")}`
|
|
1611
|
+
}
|
|
1612
|
+
|
|
1613
|
+
export interface BuildSpawnEnvOptions {
|
|
1614
|
+
repo: string
|
|
1615
|
+
/** Harness mode for the workers (e.g. "code"). */
|
|
1616
|
+
mode: string
|
|
1617
|
+
/** An explicit --model flag value, if the caller passed one — always wins (resolveModelForMode rule 1). */
|
|
1618
|
+
explicitModel?: string
|
|
1619
|
+
memoryDir?: string
|
|
1620
|
+
/** Per-session iteration cap forwarded as HEADLESSCODE_MAX_ITERATIONS (run-worker.sh → --max-iterations). */
|
|
1621
|
+
maxIterations?: number
|
|
1622
|
+
/** Issue #49: run a plan-first session per worktree before the code worker (forwarded as PLAN_FIRST=1). */
|
|
1623
|
+
planFirst?: boolean
|
|
1624
|
+
/** Mode slug for the plan-first session (default: architect). */
|
|
1625
|
+
planFirstMode?: string
|
|
1626
|
+
/** Iteration cap for the plan-first session (default: 15). */
|
|
1627
|
+
planFirstMaxIterations?: number
|
|
1628
|
+
/** Env to read OPENROUTER_MODEL from (default: process.env). */
|
|
1629
|
+
env?: NodeJS.ProcessEnv
|
|
1630
|
+
}
|
|
1631
|
+
|
|
1632
|
+
/**
|
|
1633
|
+
* Resolve the worker's model and build the env for the initial spawnSync
|
|
1634
|
+
* (step 2 of orchestrateMain). Exported for tests: this is the fix for
|
|
1635
|
+
* "mode-models.json ignored on a worker's FIRST spawn" — the resolved model
|
|
1636
|
+
* is threaded into the spawn env as OPENROUTER_MODEL (which the spawner
|
|
1637
|
+
* script and run-worker.sh inherit), instead of only applying on rework
|
|
1638
|
+
* re-spawns. The returned workerModel is the SAME const the rework loop
|
|
1639
|
+
* reuses later.
|
|
1640
|
+
*/
|
|
1641
|
+
export function buildSpawnEnv(options: BuildSpawnEnvOptions): { env: NodeJS.ProcessEnv; workerModel: string | undefined } {
|
|
1642
|
+
const workerModel = resolveModelForMode({
|
|
1643
|
+
workspaceRoot: options.repo,
|
|
1644
|
+
mode: options.mode,
|
|
1645
|
+
explicitModel: options.explicitModel,
|
|
1646
|
+
env: options.env ?? process.env,
|
|
1647
|
+
})
|
|
1648
|
+
const env: NodeJS.ProcessEnv = {
|
|
1649
|
+
...(options.env ?? process.env),
|
|
1650
|
+
TARGET_REPO: options.repo,
|
|
1651
|
+
ORCHESTRATOR_MODE: options.mode,
|
|
1652
|
+
// The resolved worker model must reach the spawner (and the workers it
|
|
1653
|
+
// launches) — same pattern as HEADLESSCODE_PROJECT below.
|
|
1654
|
+
...(workerModel ? { OPENROUTER_MODEL: workerModel } : {}),
|
|
1655
|
+
// Phase 3: workers scope memory to the REPO (not the worktree) and
|
|
1656
|
+
// share one memory dir. HEADLESSCODE_PROJECT flows to the worker CLI
|
|
1657
|
+
// which passes it as the session `project`; the env var is inherited
|
|
1658
|
+
// when --memory-dir is not given.
|
|
1659
|
+
HEADLESSCODE_PROJECT: path.basename(options.repo),
|
|
1660
|
+
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
1661
|
+
// Iteration cap: an explicit --max-iterations beats the inherited env
|
|
1662
|
+
// var; undefined leaves whatever the caller's env already had in place
|
|
1663
|
+
// (or the harness's own default, 50, if neither is set).
|
|
1664
|
+
...(options.maxIterations !== undefined ? { HEADLESSCODE_MAX_ITERATIONS: String(options.maxIterations) } : {}),
|
|
1665
|
+
// Operator knobs (spawner guardrail + local exploration) forwarded from
|
|
1666
|
+
// the AMBIENT env, not options.env — the caller's env is usually
|
|
1667
|
+
// process.env so the spread above already carries them, but a filtered
|
|
1668
|
+
// env must not silently drop an operator decision. ALLOW_UNINDEXED /
|
|
1669
|
+
// HEADLESSCODE_AUTO_INDEX gate the spawner's no-index guardrail;
|
|
1670
|
+
// HEADLESSCODE_LOCAL_EXPLORE* enable the pre-cloud local-explore phase.
|
|
1671
|
+
...(process.env.ALLOW_UNINDEXED ? { ALLOW_UNINDEXED: process.env.ALLOW_UNINDEXED } : {}),
|
|
1672
|
+
...(process.env.HEADLESSCODE_AUTO_INDEX ? { HEADLESSCODE_AUTO_INDEX: process.env.HEADLESSCODE_AUTO_INDEX } : {}),
|
|
1673
|
+
...(process.env.HEADLESSCODE_LOCAL_EXPLORE ? { HEADLESSCODE_LOCAL_EXPLORE: process.env.HEADLESSCODE_LOCAL_EXPLORE } : {}),
|
|
1674
|
+
...(process.env.HEADLESSCODE_LOCAL_EXPLORE_MODEL
|
|
1675
|
+
? { HEADLESSCODE_LOCAL_EXPLORE_MODEL: process.env.HEADLESSCODE_LOCAL_EXPLORE_MODEL }
|
|
1676
|
+
: {}),
|
|
1677
|
+
...(process.env.HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS
|
|
1678
|
+
? { HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS: process.env.HEADLESSCODE_LOCAL_EXPLORE_MAX_ITERATIONS }
|
|
1679
|
+
: {}),
|
|
1680
|
+
...(process.env.HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS
|
|
1681
|
+
? { HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS: process.env.HEADLESSCODE_LOCAL_EXPLORE_CONTEXT_TOKENS }
|
|
1682
|
+
: {}),
|
|
1683
|
+
// Issue #49 plan-first experiment: forwarded to the spawner, which runs
|
|
1684
|
+
// a short architect-mode planning session in each worktree BEFORE
|
|
1685
|
+
// launching the code worker and appends the plan (PLAN.md) to the
|
|
1686
|
+
// worker's task file. Only set when --plan-first is on — never the
|
|
1687
|
+
// default. PLAN_FIRST_MODE/MAX_ITERATIONS are only meaningful when
|
|
1688
|
+
// PLAN_FIRST is set, so they travel together.
|
|
1689
|
+
...(options.planFirst ? { PLAN_FIRST: "1" } : {}),
|
|
1690
|
+
...(options.planFirst ? { PLAN_FIRST_MODE: options.planFirstMode ?? "architect" } : {}),
|
|
1691
|
+
...(options.planFirst ? { PLAN_FIRST_MAX_ITERATIONS: String(options.planFirstMaxIterations ?? 15) } : {}),
|
|
1692
|
+
}
|
|
1693
|
+
return { env, workerModel }
|
|
1694
|
+
}
|
|
1695
|
+
|
|
1696
|
+
/**
|
|
1697
|
+
* Issue #53 pre-flight issue-size check: one loud stderr line per issue whose
|
|
1698
|
+
* body reads like 3+ independent pieces of work. Warning-only by design — a
|
|
1699
|
+
* crude heuristic can false-positive, so it never aborts the round; the
|
|
1700
|
+
* operator is told BEFORE anything spawns so they can split the issue first.
|
|
1701
|
+
*/
|
|
1702
|
+
function printIssueSizeWarnings(warnings: IssueSizeWarning[]): void {
|
|
1703
|
+
for (const w of warnings) {
|
|
1704
|
+
process.stderr.write(
|
|
1705
|
+
`headlesscode orchestrate: WARNING: issue #${w.number} ("${w.title}") reads like ${w.sections} independent pieces of work ` +
|
|
1706
|
+
`(${w.sections} top-level numbered/bulleted sections in its body). Dispatching it to ONE worker risks iteration-cap / budget ` +
|
|
1707
|
+
`burn + rework cycles (issue #53). Consider splitting it into ${w.sections} sub-issues and re-running before spawning. ` +
|
|
1708
|
+
`--no-issue-size-check silences this warning.\n`,
|
|
1709
|
+
)
|
|
1710
|
+
}
|
|
1711
|
+
}
|
|
1712
|
+
|
|
1713
|
+
function printPlan(
|
|
1714
|
+
repo: string,
|
|
1715
|
+
mode: string,
|
|
1716
|
+
specs: WorktreeSpec[],
|
|
1717
|
+
issues: SplitIssue[],
|
|
1718
|
+
models?: { worker: string; reviewer: string; qa: string },
|
|
1719
|
+
maxIterations?: number,
|
|
1720
|
+
estimateSection?: string[],
|
|
1721
|
+
/** Issue #49: plan-first session info to surface on the dry-run plan. */
|
|
1722
|
+
planFirst?: { mode: string; maxIterations: number },
|
|
1723
|
+
): void {
|
|
1724
|
+
// Shape labels come from split.ts's issueShape (the SAME taxonomy the
|
|
1725
|
+
// split and the cost estimate use) — a single source of truth instead of
|
|
1726
|
+
// a third inline copy of the keyword scans.
|
|
1727
|
+
const shapeNotes = new Map<number, string>()
|
|
1728
|
+
for (const issue of issues) {
|
|
1729
|
+
const shape = issueShape(issue)
|
|
1730
|
+
const label =
|
|
1731
|
+
shape === "hot" ? "hot-path isolated" : shape === "generic" ? "generic" : `same-shape: ${shape}`
|
|
1732
|
+
shapeNotes.set(issue.number, label)
|
|
1733
|
+
}
|
|
1734
|
+
|
|
1735
|
+
process.stdout.write("── Orchestrate plan (dry run) ─────────────────────────────\n")
|
|
1736
|
+
process.stdout.write(` repo: ${repo}\n`)
|
|
1737
|
+
process.stdout.write(` mode: ${mode}\n`)
|
|
1738
|
+
if (planFirst) {
|
|
1739
|
+
process.stdout.write(
|
|
1740
|
+
` plan-first: mode ${planFirst.mode}, max ${planFirst.maxIterations} iterations per worktree ` +
|
|
1741
|
+
`(issue #49 — plan session runs BEFORE each code worker)\n`,
|
|
1742
|
+
)
|
|
1743
|
+
}
|
|
1744
|
+
process.stdout.write(
|
|
1745
|
+
maxIterations !== undefined
|
|
1746
|
+
? ` max iterations: ${maxIterations} (workers get --max-iterations ${maxIterations})\n`
|
|
1747
|
+
: " max iterations: <harness default 250>\n",
|
|
1748
|
+
)
|
|
1749
|
+
if (models) {
|
|
1750
|
+
process.stdout.write(` models: worker: ${models.worker}\n`)
|
|
1751
|
+
process.stdout.write(` reviewer: ${models.reviewer}\n`)
|
|
1752
|
+
process.stdout.write(` QA: ${models.qa}\n`)
|
|
1753
|
+
}
|
|
1754
|
+
process.stdout.write(` issues: ${issues.map((i) => i.number).join(", ")}\n\n`)
|
|
1755
|
+
process.stdout.write("Split plan (heuristics from the multi-agent-orchestrator mode):\n")
|
|
1756
|
+
for (const spec of specs) {
|
|
1757
|
+
const reasons = spec.issues.map((n) => shapeNotes.get(n) ?? "generic").join(", ")
|
|
1758
|
+
process.stdout.write(
|
|
1759
|
+
` ${spec.name.padEnd(4)} issue ${spec.issues.join(", ").padEnd(10)} -> plans/parallel-tasks/${spec.taskFile} (${reasons})\n`,
|
|
1760
|
+
)
|
|
1761
|
+
}
|
|
1762
|
+
// Issue #16: expected cost/iteration range for this round, estimated
|
|
1763
|
+
// from recorded cost history (best-effort — a read failure yields no
|
|
1764
|
+
// section, never a failed dry-run).
|
|
1765
|
+
if (estimateSection && estimateSection.length > 0) {
|
|
1766
|
+
process.stdout.write("\nCost estimate (from recorded cost-history — issue #16):\n")
|
|
1767
|
+
for (const line of estimateSection) {
|
|
1768
|
+
process.stdout.write(` ${line}\n`)
|
|
1769
|
+
}
|
|
1770
|
+
}
|
|
1771
|
+
process.stdout.write("\nSpawn commands:\n")
|
|
1772
|
+
process.stdout.write(` ${spawnCommandFor(HARNESS_ROOT_TS, specs)}\n`)
|
|
1773
|
+
}
|
|
1774
|
+
|
|
1775
|
+
/** Resolve this harness repo's root (parent of src/orchestrator). */
|
|
1776
|
+
const HARNESS_ROOT_TS = fileURLToPath(new URL("../..", import.meta.url))
|
|
1777
|
+
|
|
1778
|
+
// ─── `orchestrate status` — read-side status / wait for EXTERNAL callers ────
|
|
1779
|
+
|
|
1780
|
+
const STATUS_USAGE = `headlesscode orchestrate status — per-group status of a round, optionally waiting for it
|
|
1781
|
+
|
|
1782
|
+
Usage:
|
|
1783
|
+
headlesscode orchestrate status --repo <path> [--json]
|
|
1784
|
+
headlesscode orchestrate status --repo <path> --wait [--timeout-ms <n>] [--on-group-terminal <cmd>] [--json]
|
|
1785
|
+
|
|
1786
|
+
Options:
|
|
1787
|
+
--repo <path> Target repo root whose .worktrees/.orchestrator-state.json
|
|
1788
|
+
is read (required). Before reporting, stale non-terminal
|
|
1789
|
+
entries are reconciled against real worktree markers
|
|
1790
|
+
(.harness.done/.harness.exit) and the state file is
|
|
1791
|
+
patched back when ground truth proves a different status
|
|
1792
|
+
("done"/"failed", or "orphaned" when the worktree is gone).
|
|
1793
|
+
--wait Block inside this ONE call until every group reaches a
|
|
1794
|
+
terminal status (done|failed|needs-human|orphaned;
|
|
1795
|
+
"blocked" is NOT terminal — it keeps waiting) or
|
|
1796
|
+
--timeout-ms elapses, then print the same summary plus a
|
|
1797
|
+
one-line verdict. Reconciliation runs on every poll.
|
|
1798
|
+
--timeout-ms <n> Max --wait duration, ms (default: 7200000 = 2h, matching
|
|
1799
|
+
the watcher's stall guard — a round left non-terminal
|
|
1800
|
+
longer than that is flagged stalled anyway).
|
|
1801
|
+
--poll-interval-ms <n> State-file poll interval for --wait, ms (default: 5000,
|
|
1802
|
+
matching the watcher's own write cadence).
|
|
1803
|
+
--on-group-terminal <cmd>
|
|
1804
|
+
With --wait: run <cmd> (spawned directly — no shell
|
|
1805
|
+
parsing — with the group name and its terminal status as
|
|
1806
|
+
two positional args) each time a group transitions to
|
|
1807
|
+
done|failed|needs-human|orphaned mid-wait. Fires once per
|
|
1808
|
+
group per transition; never fires for groups already
|
|
1809
|
+
terminal when the wait started. A failing hook logs a
|
|
1810
|
+
warning to stderr but never aborts the wait.
|
|
1811
|
+
--json Machine-readable output. One-shot: the reconciled
|
|
1812
|
+
OrchestratorState plus a top-level "reconciled" array of
|
|
1813
|
+
group names patched (only when something was patched).
|
|
1814
|
+
With --wait: the final state plus top-level
|
|
1815
|
+
allDone/timedOut/verdict/reconciled fields.
|
|
1816
|
+
--help, -h Show this help and exit.
|
|
1817
|
+
|
|
1818
|
+
Exit codes:
|
|
1819
|
+
0 one-shot: no group is failed/needs-human/orphaned right now;
|
|
1820
|
+
--wait: every group is done (clean finish)
|
|
1821
|
+
1 one-shot: at least one group is failed, needs-human, or orphaned;
|
|
1822
|
+
--wait: any group failed/needs-human/orphaned, or the wait timed out
|
|
1823
|
+
2 usage error (bad or missing arguments)
|
|
1824
|
+
`
|
|
1825
|
+
|
|
1826
|
+
export interface StatusOptions {
|
|
1827
|
+
repo: string
|
|
1828
|
+
json: boolean
|
|
1829
|
+
wait: boolean
|
|
1830
|
+
timeoutMs: number
|
|
1831
|
+
pollIntervalMs: number
|
|
1832
|
+
/** Command spawned (with the group name + terminal status as args) per terminal transition while --wait is active. */
|
|
1833
|
+
onGroupTerminal?: string
|
|
1834
|
+
help: boolean
|
|
1835
|
+
}
|
|
1836
|
+
|
|
1837
|
+
export function parseStatusArgs(argv: string[]): { options: StatusOptions; error?: string } {
|
|
1838
|
+
const options: StatusOptions = {
|
|
1839
|
+
repo: "",
|
|
1840
|
+
json: false,
|
|
1841
|
+
wait: false,
|
|
1842
|
+
timeoutMs: DEFAULT_STATUS_TIMEOUT_MS,
|
|
1843
|
+
pollIntervalMs: DEFAULT_STATUS_POLL_INTERVAL_MS,
|
|
1844
|
+
onGroupTerminal: undefined,
|
|
1845
|
+
help: false,
|
|
1846
|
+
}
|
|
1847
|
+
for (let i = 0; i < argv.length; i++) {
|
|
1848
|
+
const arg = argv[i]
|
|
1849
|
+
const eq = arg.indexOf("=")
|
|
1850
|
+
const flag = eq === -1 ? arg : arg.slice(0, eq)
|
|
1851
|
+
const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
|
|
1852
|
+
const next = (): string | undefined => {
|
|
1853
|
+
if (inlineValue !== undefined) {
|
|
1854
|
+
return inlineValue
|
|
1855
|
+
}
|
|
1856
|
+
const v = argv[i + 1]
|
|
1857
|
+
if (v === undefined || v.startsWith("--")) {
|
|
1858
|
+
return undefined
|
|
1859
|
+
}
|
|
1860
|
+
i++
|
|
1861
|
+
return v
|
|
1862
|
+
}
|
|
1863
|
+
switch (flag) {
|
|
1864
|
+
case "--repo": {
|
|
1865
|
+
const v = next()
|
|
1866
|
+
if (v === undefined) {
|
|
1867
|
+
return { options, error: "Missing value for --repo" }
|
|
1868
|
+
}
|
|
1869
|
+
options.repo = v
|
|
1870
|
+
break
|
|
1871
|
+
}
|
|
1872
|
+
case "--json":
|
|
1873
|
+
options.json = true
|
|
1874
|
+
break
|
|
1875
|
+
case "--wait":
|
|
1876
|
+
options.wait = true
|
|
1877
|
+
break
|
|
1878
|
+
case "--timeout-ms": {
|
|
1879
|
+
const v = next()
|
|
1880
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
1881
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
1882
|
+
return { options, error: "--timeout-ms requires a positive integer" }
|
|
1883
|
+
}
|
|
1884
|
+
options.timeoutMs = n
|
|
1885
|
+
break
|
|
1886
|
+
}
|
|
1887
|
+
case "--poll-interval-ms": {
|
|
1888
|
+
const v = next()
|
|
1889
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
1890
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
1891
|
+
return { options, error: "--poll-interval-ms requires a positive integer" }
|
|
1892
|
+
}
|
|
1893
|
+
options.pollIntervalMs = n
|
|
1894
|
+
break
|
|
1895
|
+
}
|
|
1896
|
+
case "--on-group-terminal": {
|
|
1897
|
+
const v = next()
|
|
1898
|
+
if (v === undefined) {
|
|
1899
|
+
return { options, error: "Missing value for --on-group-terminal" }
|
|
1900
|
+
}
|
|
1901
|
+
options.onGroupTerminal = v
|
|
1902
|
+
break
|
|
1903
|
+
}
|
|
1904
|
+
case "--help":
|
|
1905
|
+
case "-h":
|
|
1906
|
+
options.help = true
|
|
1907
|
+
break
|
|
1908
|
+
default:
|
|
1909
|
+
return { options, error: `Unknown orchestrate status argument: ${arg}` }
|
|
1910
|
+
}
|
|
1911
|
+
}
|
|
1912
|
+
return { options }
|
|
1913
|
+
}
|
|
1914
|
+
|
|
1915
|
+
export interface StatusIo {
|
|
1916
|
+
stdout?: (text: string) => void
|
|
1917
|
+
stderr?: (text: string) => void
|
|
1918
|
+
}
|
|
1919
|
+
|
|
1920
|
+
function waitExitCode(allDone: boolean, summary: StatusSummary): number {
|
|
1921
|
+
// Nothing in the state file is not a failure — exit 0 so a caller that
|
|
1922
|
+
// ran --wait against a wrong/empty path sees a clear, non-alarming result.
|
|
1923
|
+
if (summary.counts.total === 0) {
|
|
1924
|
+
return 0
|
|
1925
|
+
}
|
|
1926
|
+
if (allDone) {
|
|
1927
|
+
return 0
|
|
1928
|
+
}
|
|
1929
|
+
// Reached here only when allDone is false: either the wait timed out with
|
|
1930
|
+
// groups still non-terminal, or every group is terminal but at least one
|
|
1931
|
+
// failed/needs-human. Both are non-clean outcomes → 1.
|
|
1932
|
+
return 1
|
|
1933
|
+
}
|
|
1934
|
+
|
|
1935
|
+
/**
|
|
1936
|
+
* Run the `--on-group-terminal <cmd>` hook for a group that just reached a
|
|
1937
|
+
* terminal status: spawn <cmd> directly (NO shell parsing of <cmd> — it is
|
|
1938
|
+
* the executable path/name) with the group name and status as the two
|
|
1939
|
+
* positional args, and await completion so transitions fire sequentially.
|
|
1940
|
+
* A broken hook (spawn failure or non-zero exit) logs a warning to stderr
|
|
1941
|
+
* but never aborts the wait — a failed notification must not take down a
|
|
1942
|
+
* status watch that is otherwise healthy.
|
|
1943
|
+
*/
|
|
1944
|
+
async function runGroupTerminalHook(cmd: string, group: OrchestratorGroup, writeErr: (text: string) => void): Promise<void> {
|
|
1945
|
+
try {
|
|
1946
|
+
const child = spawn(cmd, [group.name, group.status], { stdio: "inherit" })
|
|
1947
|
+
await new Promise<void>((resolve, reject) => {
|
|
1948
|
+
child.once("error", reject)
|
|
1949
|
+
child.once("exit", (code, signal) => {
|
|
1950
|
+
if (code !== 0) {
|
|
1951
|
+
reject(new Error(signal ? `killed by ${signal}` : `exit ${code}`))
|
|
1952
|
+
} else {
|
|
1953
|
+
resolve()
|
|
1954
|
+
}
|
|
1955
|
+
})
|
|
1956
|
+
})
|
|
1957
|
+
} catch (err) {
|
|
1958
|
+
writeErr(
|
|
1959
|
+
`headlesscode orchestrate status: --on-group-terminal hook failed for group "${group.name}" ` +
|
|
1960
|
+
`(${group.status}): ${err instanceof Error ? err.message : String(err)}\n`,
|
|
1961
|
+
)
|
|
1962
|
+
}
|
|
1963
|
+
}
|
|
1964
|
+
|
|
1965
|
+
/**
|
|
1966
|
+
* Attach the cleanup-eligibility column to every TERMINAL group's status row
|
|
1967
|
+
* (read-only, no side effects — the same gates the cleanup command uses,
|
|
1968
|
+
* minus the GitHub PR network check). Returns the per-group map for --json.
|
|
1969
|
+
* Never throws: on any failure the group is reported blocked with a reason,
|
|
1970
|
+
* matching cleanup's fail-closed model.
|
|
1971
|
+
*/
|
|
1972
|
+
function attachCleanupStatus(repo: string, state: OrchestratorState, summary: StatusSummary): Record<string, CleanupStatus> {
|
|
1973
|
+
let baseBranch: string | undefined
|
|
1974
|
+
try {
|
|
1975
|
+
baseBranch = resolveBaseBranch(repo)
|
|
1976
|
+
} catch {
|
|
1977
|
+
baseBranch = undefined
|
|
1978
|
+
}
|
|
1979
|
+
const byGroup: Record<string, CleanupStatus> = {}
|
|
1980
|
+
for (const row of summary.groups) {
|
|
1981
|
+
if (!isTerminalStatus(row.status)) {
|
|
1982
|
+
continue // non-terminal groups are never touched by cleanup
|
|
1983
|
+
}
|
|
1984
|
+
const group = state.groups.find((g) => g.name === row.name)
|
|
1985
|
+
if (!group) {
|
|
1986
|
+
continue
|
|
1987
|
+
}
|
|
1988
|
+
const cleanup: CleanupStatus =
|
|
1989
|
+
baseBranch === undefined
|
|
1990
|
+
? { status: "blocked", reason: "cannot resolve base branch (detached HEAD?) — pass --base to cleanup" }
|
|
1991
|
+
: assessGroupCleanupSync(repo, group, baseBranch)
|
|
1992
|
+
row.cleanup = cleanup
|
|
1993
|
+
byGroup[row.name] = cleanup
|
|
1994
|
+
}
|
|
1995
|
+
return byGroup
|
|
1996
|
+
}
|
|
1997
|
+
|
|
1998
|
+
/**
|
|
1999
|
+
* `headlesscode orchestrate status` entry point. One call, one result: the
|
|
2000
|
+
* calling agent never re-invokes a tool per poll, unlike the bash loop this
|
|
2001
|
+
* replaces. Both forms reconcile stale state against real worktree markers
|
|
2002
|
+
* before reporting (see status.ts's reconcileGroups) — the state file is
|
|
2003
|
+
* patched back when ground truth proves a different status, so a stale
|
|
2004
|
+
* "running" entry is never reported as-is.
|
|
2005
|
+
*/
|
|
2006
|
+
export async function statusMain(argv: string[], io: StatusIo = {}): Promise<number> {
|
|
2007
|
+
const writeOut = io.stdout ?? ((text: string) => process.stdout.write(text))
|
|
2008
|
+
const writeErr = io.stderr ?? ((text: string) => process.stderr.write(text))
|
|
2009
|
+
|
|
2010
|
+
const { options, error } = parseStatusArgs(argv)
|
|
2011
|
+
if (error) {
|
|
2012
|
+
writeErr(`headlesscode orchestrate status: ${error}\n\n${STATUS_USAGE}`)
|
|
2013
|
+
return 2
|
|
2014
|
+
}
|
|
2015
|
+
if (options.help) {
|
|
2016
|
+
writeOut(STATUS_USAGE)
|
|
2017
|
+
return 0
|
|
2018
|
+
}
|
|
2019
|
+
if (!options.repo) {
|
|
2020
|
+
writeErr(`headlesscode orchestrate status: --repo <path> is required\n\n${STATUS_USAGE}`)
|
|
2021
|
+
return 2
|
|
2022
|
+
}
|
|
2023
|
+
|
|
2024
|
+
const repo = path.resolve(options.repo)
|
|
2025
|
+
const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
|
|
2026
|
+
|
|
2027
|
+
if (options.wait) {
|
|
2028
|
+
// --on-group-terminal: spawn the hook command per genuine terminal
|
|
2029
|
+
// transition (sequentially — terminal transitions are rare and
|
|
2030
|
+
// blocking the poll loop for the hook is fine).
|
|
2031
|
+
const hookCmd = options.onGroupTerminal
|
|
2032
|
+
const result = await waitForTerminalState(statePath, {
|
|
2033
|
+
timeoutMs: options.timeoutMs,
|
|
2034
|
+
pollIntervalMs: options.pollIntervalMs,
|
|
2035
|
+
// Reconcile on EVERY poll (a group can finish mid-wait). The
|
|
2036
|
+
// hook reports the patch; waitForTerminalState persists it ONLY
|
|
2037
|
+
// when no live watcher owns the state file (issue #116: a
|
|
2038
|
+
// concurrent `status --wait` persisting its read-side reconcile
|
|
2039
|
+
// pre-empted the watcher, which then short-circuited the group
|
|
2040
|
+
// and silently skipped log analysis + the automated review —
|
|
2041
|
+
// see waitForTerminalState's live-watcher heartbeat check).
|
|
2042
|
+
reconcile: (state) => reconcileGroups(state, repo),
|
|
2043
|
+
onGroupTerminal: hookCmd ? (group) => runGroupTerminalHook(hookCmd, group, writeErr) : undefined,
|
|
2044
|
+
})
|
|
2045
|
+
const summary = buildStatusSummary(result.state)
|
|
2046
|
+
const verdict = verdictLine(summary, result.timedOut, result.allDone)
|
|
2047
|
+
const cleanupByGroup = attachCleanupStatus(repo, result.state, summary)
|
|
2048
|
+
if (options.json) {
|
|
2049
|
+
const out: Record<string, unknown> = {
|
|
2050
|
+
...result.state,
|
|
2051
|
+
allDone: result.allDone,
|
|
2052
|
+
timedOut: result.timedOut,
|
|
2053
|
+
verdict,
|
|
2054
|
+
}
|
|
2055
|
+
if (result.reconciled.length > 0) {
|
|
2056
|
+
out.reconciled = result.reconciled
|
|
2057
|
+
}
|
|
2058
|
+
if (Object.keys(cleanupByGroup).length > 0) {
|
|
2059
|
+
out.cleanup = cleanupByGroup
|
|
2060
|
+
}
|
|
2061
|
+
writeOut(JSON.stringify(out, null, 2) + "\n")
|
|
2062
|
+
} else {
|
|
2063
|
+
writeOut(
|
|
2064
|
+
formatStatusText(summary, {
|
|
2065
|
+
repo,
|
|
2066
|
+
statePath,
|
|
2067
|
+
timedOut: result.timedOut,
|
|
2068
|
+
allDone: result.allDone,
|
|
2069
|
+
elapsedMs: result.elapsedMs,
|
|
2070
|
+
reconciled: result.reconciled,
|
|
2071
|
+
}),
|
|
2072
|
+
)
|
|
2073
|
+
}
|
|
2074
|
+
return waitExitCode(result.allDone, summary)
|
|
2075
|
+
}
|
|
2076
|
+
|
|
2077
|
+
// One-shot: read the current state, reconcile it against real markers
|
|
2078
|
+
// (persisting any patch), print, exit on current health.
|
|
2079
|
+
let state: OrchestratorState
|
|
2080
|
+
try {
|
|
2081
|
+
state = loadStateSync(statePath)
|
|
2082
|
+
} catch (err) {
|
|
2083
|
+
writeErr(
|
|
2084
|
+
`headlesscode orchestrate status: cannot read state file ${statePath}: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
2085
|
+
)
|
|
2086
|
+
return 1
|
|
2087
|
+
}
|
|
2088
|
+
const reconciliation = reconcileGroups(state, repo)
|
|
2089
|
+
if (reconciliation.reconciled.length > 0) {
|
|
2090
|
+
saveStateSync(statePath, reconciliation.state)
|
|
2091
|
+
state = reconciliation.state
|
|
2092
|
+
}
|
|
2093
|
+
const summary = buildStatusSummary(state)
|
|
2094
|
+
const cleanupByGroup = attachCleanupStatus(repo, state, summary)
|
|
2095
|
+
if (options.json) {
|
|
2096
|
+
const out: Record<string, unknown> = { ...state }
|
|
2097
|
+
if (reconciliation.reconciled.length > 0) {
|
|
2098
|
+
out.reconciled = reconciliation.reconciled
|
|
2099
|
+
}
|
|
2100
|
+
if (Object.keys(cleanupByGroup).length > 0) {
|
|
2101
|
+
out.cleanup = cleanupByGroup
|
|
2102
|
+
}
|
|
2103
|
+
writeOut(JSON.stringify(out, null, 2) + "\n")
|
|
2104
|
+
} else {
|
|
2105
|
+
writeOut(formatStatusText(summary, { repo, statePath, reconciled: reconciliation.reconciled }))
|
|
2106
|
+
}
|
|
2107
|
+
return summary.counts.failed > 0 || summary.counts.needsHuman > 0 || summary.counts.orphaned > 0 ? 1 : 0
|
|
2108
|
+
}
|
|
2109
|
+
|
|
2110
|
+
// ─── `orchestrate stop` — stop a group's WHOLE worker tree (issue #20) ──────
|
|
2111
|
+
|
|
2112
|
+
const STOP_USAGE = `headlesscode orchestrate stop — stop one or more groups' worker process trees
|
|
2113
|
+
|
|
2114
|
+
Usage:
|
|
2115
|
+
headlesscode orchestrate stop --repo <path> --group <name> [--group <name> ...] [options]
|
|
2116
|
+
|
|
2117
|
+
Stops a worker COMPLETELY: scripts/stop-worker.sh targets the worker's process
|
|
2118
|
+
GROUP — run-worker.sh launches each worker under setsid and the wrapper
|
|
2119
|
+
records its process-group id in <worktree>/.harness.pgid — so the wrapper
|
|
2120
|
+
bash, npx, node/tsx and every non-detached grandchild are all killed.
|
|
2121
|
+
Killing the wrapper PID alone used to leave the real work running undetected,
|
|
2122
|
+
reparented to init, until the session finished on its own (issue #20).
|
|
2123
|
+
|
|
2124
|
+
Each stopped group is patched to "needs-human" (a TERMINAL state — a human
|
|
2125
|
+
must decide what to do with the stopped worktree next, e.g. resume it under
|
|
2126
|
+
new logic). Groups that are already terminal are left untouched.
|
|
2127
|
+
|
|
2128
|
+
Options:
|
|
2129
|
+
--repo <path> Target repo root whose .worktrees/.orchestrator-state.json
|
|
2130
|
+
is read (required).
|
|
2131
|
+
--group <name> Worktree group name to stop (repeatable; at least one).
|
|
2132
|
+
--grace-ms <n> SIGTERM grace period before SIGKILL escalation, ms
|
|
2133
|
+
(default: 5000; forwarded to scripts/stop-worker.sh).
|
|
2134
|
+
--json Machine-readable output: per-group results + final state.
|
|
2135
|
+
--help, -h Show this help and exit.
|
|
2136
|
+
|
|
2137
|
+
Exit codes:
|
|
2138
|
+
0 every requested group was stopped (or was already terminal)
|
|
2139
|
+
1 a requested group could not be stopped / is not in the state file
|
|
2140
|
+
2 usage error (bad or missing arguments)
|
|
2141
|
+
`
|
|
2142
|
+
|
|
2143
|
+
export interface StopOptions {
|
|
2144
|
+
repo: string
|
|
2145
|
+
groups: string[]
|
|
2146
|
+
graceMs: number
|
|
2147
|
+
json: boolean
|
|
2148
|
+
help: boolean
|
|
2149
|
+
}
|
|
2150
|
+
|
|
2151
|
+
export function parseStopArgs(argv: string[]): { options: StopOptions; error?: string } {
|
|
2152
|
+
const options: StopOptions = { repo: "", groups: [], graceMs: 5000, json: false, help: false }
|
|
2153
|
+
for (let i = 0; i < argv.length; i++) {
|
|
2154
|
+
const arg = argv[i]
|
|
2155
|
+
const eq = arg.indexOf("=")
|
|
2156
|
+
const flag = eq === -1 ? arg : arg.slice(0, eq)
|
|
2157
|
+
const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
|
|
2158
|
+
const next = (): string | undefined => {
|
|
2159
|
+
if (inlineValue !== undefined) {
|
|
2160
|
+
return inlineValue
|
|
2161
|
+
}
|
|
2162
|
+
const v = argv[i + 1]
|
|
2163
|
+
if (v === undefined || v.startsWith("--")) {
|
|
2164
|
+
return undefined
|
|
2165
|
+
}
|
|
2166
|
+
i++
|
|
2167
|
+
return v
|
|
2168
|
+
}
|
|
2169
|
+
switch (flag) {
|
|
2170
|
+
case "--repo": {
|
|
2171
|
+
const v = next()
|
|
2172
|
+
if (v === undefined) {
|
|
2173
|
+
return { options, error: "Missing value for --repo" }
|
|
2174
|
+
}
|
|
2175
|
+
options.repo = v
|
|
2176
|
+
break
|
|
2177
|
+
}
|
|
2178
|
+
case "--group": {
|
|
2179
|
+
const v = next()
|
|
2180
|
+
if (v === undefined) {
|
|
2181
|
+
return { options, error: "Missing value for --group" }
|
|
2182
|
+
}
|
|
2183
|
+
options.groups.push(v)
|
|
2184
|
+
break
|
|
2185
|
+
}
|
|
2186
|
+
case "--grace-ms": {
|
|
2187
|
+
const v = next()
|
|
2188
|
+
const n = v === undefined ? Number.NaN : Number(v)
|
|
2189
|
+
if (!Number.isInteger(n) || n < 0) {
|
|
2190
|
+
return { options, error: "--grace-ms requires a non-negative integer" }
|
|
2191
|
+
}
|
|
2192
|
+
options.graceMs = n
|
|
2193
|
+
break
|
|
2194
|
+
}
|
|
2195
|
+
case "--json":
|
|
2196
|
+
options.json = true
|
|
2197
|
+
break
|
|
2198
|
+
case "--help":
|
|
2199
|
+
case "-h":
|
|
2200
|
+
options.help = true
|
|
2201
|
+
break
|
|
2202
|
+
default:
|
|
2203
|
+
return { options, error: `Unknown orchestrate stop argument: ${arg}` }
|
|
2204
|
+
}
|
|
2205
|
+
}
|
|
2206
|
+
return { options }
|
|
2207
|
+
}
|
|
2208
|
+
|
|
2209
|
+
export interface StopIo {
|
|
2210
|
+
stdout?: (text: string) => void
|
|
2211
|
+
stderr?: (text: string) => void
|
|
2212
|
+
/** Injectable stop-script runner (tests); default: spawnSync scripts/stop-worker.sh. */
|
|
2213
|
+
runStopWorker?: (wtPath: string, graceMs: number) => { exitCode: number; output: string }
|
|
2214
|
+
}
|
|
2215
|
+
|
|
2216
|
+
/**
|
|
2217
|
+
* `headlesscode orchestrate stop` entry point. Stops each requested group's
|
|
2218
|
+
* worker process tree via scripts/stop-worker.sh (issue #20) and patches the
|
|
2219
|
+
* stopped groups to `needs-human` so they never sit "running" forever with no
|
|
2220
|
+
* process behind them.
|
|
2221
|
+
*/
|
|
2222
|
+
export async function stopMain(argv: string[], io: StopIo = {}): Promise<number> {
|
|
2223
|
+
const writeOut = io.stdout ?? ((text: string) => process.stdout.write(text))
|
|
2224
|
+
const writeErr = io.stderr ?? ((text: string) => process.stderr.write(text))
|
|
2225
|
+
|
|
2226
|
+
const { options, error } = parseStopArgs(argv)
|
|
2227
|
+
if (error) {
|
|
2228
|
+
writeErr(`headlesscode orchestrate stop: ${error}\n\n${STOP_USAGE}`)
|
|
2229
|
+
return 2
|
|
2230
|
+
}
|
|
2231
|
+
if (options.help) {
|
|
2232
|
+
writeOut(STOP_USAGE)
|
|
2233
|
+
return 0
|
|
2234
|
+
}
|
|
2235
|
+
if (!options.repo) {
|
|
2236
|
+
writeErr(`headlesscode orchestrate stop: --repo <path> is required\n\n${STOP_USAGE}`)
|
|
2237
|
+
return 2
|
|
2238
|
+
}
|
|
2239
|
+
if (options.groups.length === 0) {
|
|
2240
|
+
writeErr(`headlesscode orchestrate stop: at least one --group <name> is required\n\n${STOP_USAGE}`)
|
|
2241
|
+
return 2
|
|
2242
|
+
}
|
|
2243
|
+
|
|
2244
|
+
const repo = path.resolve(options.repo)
|
|
2245
|
+
const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
|
|
2246
|
+
const runStopWorker =
|
|
2247
|
+
io.runStopWorker ??
|
|
2248
|
+
((wtPath: string, graceMs: number): { exitCode: number; output: string } => {
|
|
2249
|
+
const res = spawnSync(
|
|
2250
|
+
"bash",
|
|
2251
|
+
[path.join(HARNESS_ROOT_TS, "scripts", "stop-worker.sh"), wtPath, "--grace-ms", String(graceMs)],
|
|
2252
|
+
{ encoding: "utf-8" },
|
|
2253
|
+
)
|
|
2254
|
+
const output = `${res.stdout ?? ""}${res.stderr ?? ""}`.trim()
|
|
2255
|
+
return { exitCode: res.status ?? 1, output }
|
|
2256
|
+
})
|
|
2257
|
+
|
|
2258
|
+
if (!fs.existsSync(statePath)) {
|
|
2259
|
+
writeErr(`headlesscode orchestrate stop: no orchestrator state file at ${statePath} — nothing to stop (wrong --repo?)\n`)
|
|
2260
|
+
return 1
|
|
2261
|
+
}
|
|
2262
|
+
|
|
2263
|
+
let state: OrchestratorState
|
|
2264
|
+
try {
|
|
2265
|
+
state = loadStateSync(statePath)
|
|
2266
|
+
} catch (err) {
|
|
2267
|
+
writeErr(
|
|
2268
|
+
`headlesscode orchestrate stop: cannot read state file ${statePath}: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
2269
|
+
)
|
|
2270
|
+
return 1
|
|
2271
|
+
}
|
|
2272
|
+
|
|
2273
|
+
// Validate EVERY requested group exists BEFORE stopping anything — an
|
|
2274
|
+
// operator typo must not stop half the requested set.
|
|
2275
|
+
const unknown = options.groups.filter((name) => !state.groups.some((g) => g.name === name))
|
|
2276
|
+
if (unknown.length > 0) {
|
|
2277
|
+
writeErr(
|
|
2278
|
+
`headlesscode orchestrate stop: group(s) not in state file ${statePath}: ${unknown.join(", ")}\n`,
|
|
2279
|
+
)
|
|
2280
|
+
return 1
|
|
2281
|
+
}
|
|
2282
|
+
|
|
2283
|
+
const results: Array<{ name: string; status: string; stopped: boolean; output?: string }> = []
|
|
2284
|
+
let failed = 0
|
|
2285
|
+
let current = state
|
|
2286
|
+
for (const name of options.groups) {
|
|
2287
|
+
const group = current.groups.find((g) => g.name === name)
|
|
2288
|
+
if (!group) {
|
|
2289
|
+
continue
|
|
2290
|
+
}
|
|
2291
|
+
// With --json, stdout stays pure JSON — human progress + the stop
|
|
2292
|
+
// script's own output go to stderr (conventional diagnostics channel).
|
|
2293
|
+
const writeDiag = options.json ? writeErr : writeOut
|
|
2294
|
+
if (isTerminalStatus(group.status)) {
|
|
2295
|
+
// Already done/failed/needs-human/orphaned: no live worker to stop,
|
|
2296
|
+
// and patching it would clobber the real outcome.
|
|
2297
|
+
writeDiag(`[stop] ${name}: already ${group.status} — nothing to stop\n`)
|
|
2298
|
+
results.push({ name, status: group.status, stopped: false })
|
|
2299
|
+
continue
|
|
2300
|
+
}
|
|
2301
|
+
const wtPath = groupWorktreePath(repo, group)
|
|
2302
|
+
const res = runStopWorker(wtPath, options.graceMs)
|
|
2303
|
+
if (res.exitCode !== 0) {
|
|
2304
|
+
writeDiag(res.output ? `${res.output}\n` : "")
|
|
2305
|
+
writeErr(
|
|
2306
|
+
`[stop] ${name}: scripts/stop-worker.sh failed (exit ${res.exitCode}) — group left as-is (${group.status})\n`,
|
|
2307
|
+
)
|
|
2308
|
+
failed++
|
|
2309
|
+
results.push({ name, status: group.status, stopped: false, ...(res.output ? { output: res.output } : {}) })
|
|
2310
|
+
continue
|
|
2311
|
+
}
|
|
2312
|
+
current = updateGroup(current, name, {
|
|
2313
|
+
status: "needs-human",
|
|
2314
|
+
stopped_at: new Date().toISOString(),
|
|
2315
|
+
stalled: undefined,
|
|
2316
|
+
blocked: undefined,
|
|
2317
|
+
last_activity: {
|
|
2318
|
+
note: "stopped by operator: worker process tree killed (scripts/stop-worker.sh); a human must decide next steps",
|
|
2319
|
+
},
|
|
2320
|
+
})
|
|
2321
|
+
results.push({ name, status: "needs-human", stopped: true })
|
|
2322
|
+
writeDiag(res.output ? `${res.output}\n` : "")
|
|
2323
|
+
writeDiag(`[stop] ${name}: worker tree stopped; group marked needs-human\n`)
|
|
2324
|
+
}
|
|
2325
|
+
saveStateSync(statePath, current)
|
|
2326
|
+
|
|
2327
|
+
if (options.json) {
|
|
2328
|
+
const out: Record<string, unknown> = {
|
|
2329
|
+
results,
|
|
2330
|
+
stopped: results.filter((r) => r.stopped).map((r) => r.name),
|
|
2331
|
+
groups: current.groups,
|
|
2332
|
+
}
|
|
2333
|
+
writeOut(JSON.stringify(out, null, 2) + "\n")
|
|
2334
|
+
}
|
|
2335
|
+
return failed > 0 ? 1 : 0
|
|
2336
|
+
}
|
|
2337
|
+
|
|
2338
|
+
// ─── `pipeline` subcommand (issue #148) ───────────────────────────────────────
|
|
2339
|
+
|
|
2340
|
+
const PIPELINE_USAGE = `headlesscode pipeline — run the stage-isolated research→filing pipeline
|
|
2341
|
+
|
|
2342
|
+
Usage:
|
|
2343
|
+
headlesscode pipeline --workspace <path> [--research-task <text>] [--skip-filing]
|
|
2344
|
+
[--model <id>] [--max-iterations <n>]
|
|
2345
|
+
|
|
2346
|
+
Runs the stage-isolated pipeline from issue #148: a fresh \`researcher\`-mode
|
|
2347
|
+
session produces ONE well-cited research doc on disk (the artifact gate is
|
|
2348
|
+
bound by default), then a fresh \`issue-filer\`-mode session turns that doc's
|
|
2349
|
+
CONTENT (read off disk — never the prior session's conversation) into real
|
|
2350
|
+
GitHub issue(s). Each stage is a genuinely separate \`HeadlessSession\` with
|
|
2351
|
+
its own context, mode, and executor.
|
|
2352
|
+
|
|
2353
|
+
Options:
|
|
2354
|
+
--workspace <path> Workspace root the stages run against (required; must be
|
|
2355
|
+
a git repo — the filer needs \`gh\` against its origin)
|
|
2356
|
+
--research-task <t> Task text for the research stage (default: a built-in
|
|
2357
|
+
prompt naming the artifact pattern + required sections)
|
|
2358
|
+
--skip-filing Stop after the research stage; do NOT run the filing stage
|
|
2359
|
+
--model <id> Model override for both stages (default: env / client)
|
|
2360
|
+
--max-iterations <n> Per-stage iteration cap (default: 60)
|
|
2361
|
+
--help Show this help and exit
|
|
2362
|
+
|
|
2363
|
+
Exit codes:
|
|
2364
|
+
0 all run stages succeeded
|
|
2365
|
+
1 a run stage failed (research session error / no artifact / filing parse error)
|
|
2366
|
+
2 usage error (missing --workspace, unknown flag)
|
|
2367
|
+
`
|
|
2368
|
+
|
|
2369
|
+
interface PipelineCliOptions {
|
|
2370
|
+
workspace?: string
|
|
2371
|
+
researchTask?: string
|
|
2372
|
+
skipFiling: boolean
|
|
2373
|
+
model?: string
|
|
2374
|
+
maxIterations?: number
|
|
2375
|
+
help: boolean
|
|
2376
|
+
}
|
|
2377
|
+
|
|
2378
|
+
export function parsePipelineArgs(argv: string[]): { options: PipelineCliOptions; error?: string } {
|
|
2379
|
+
const options: PipelineCliOptions = { skipFiling: false, help: false }
|
|
2380
|
+
for (let i = 0; i < argv.length; i++) {
|
|
2381
|
+
const arg = argv[i]
|
|
2382
|
+
const eq = arg.indexOf("=")
|
|
2383
|
+
const flag = eq === -1 ? arg : arg.slice(0, eq)
|
|
2384
|
+
const inlineValue = eq === -1 ? undefined : arg.slice(eq + 1)
|
|
2385
|
+
const next = (): string | undefined => {
|
|
2386
|
+
if (inlineValue !== undefined) {
|
|
2387
|
+
return inlineValue
|
|
2388
|
+
}
|
|
2389
|
+
const v = argv[i + 1]
|
|
2390
|
+
if (v === undefined || v.startsWith("--")) {
|
|
2391
|
+
return undefined
|
|
2392
|
+
}
|
|
2393
|
+
i++
|
|
2394
|
+
return v
|
|
2395
|
+
}
|
|
2396
|
+
switch (flag) {
|
|
2397
|
+
case "--workspace":
|
|
2398
|
+
case "--research-task":
|
|
2399
|
+
case "--model":
|
|
2400
|
+
case "--max-iterations": {
|
|
2401
|
+
const value = next()
|
|
2402
|
+
if (value === undefined) {
|
|
2403
|
+
return { options, error: `Missing value for ${flag}` }
|
|
2404
|
+
}
|
|
2405
|
+
if (flag === "--workspace") {
|
|
2406
|
+
options.workspace = value
|
|
2407
|
+
} else if (flag === "--research-task") {
|
|
2408
|
+
options.researchTask = value
|
|
2409
|
+
} else if (flag === "--model") {
|
|
2410
|
+
options.model = value
|
|
2411
|
+
} else {
|
|
2412
|
+
const n = Number(value)
|
|
2413
|
+
if (!Number.isInteger(n) || n <= 0) {
|
|
2414
|
+
return { options, error: "--max-iterations requires a positive integer" }
|
|
2415
|
+
}
|
|
2416
|
+
options.maxIterations = n
|
|
2417
|
+
}
|
|
2418
|
+
break
|
|
2419
|
+
}
|
|
2420
|
+
case "--skip-filing":
|
|
2421
|
+
options.skipFiling = true
|
|
2422
|
+
break
|
|
2423
|
+
case "--help":
|
|
2424
|
+
case "-h":
|
|
2425
|
+
options.help = true
|
|
2426
|
+
break
|
|
2427
|
+
default:
|
|
2428
|
+
return { options, error: `Unknown argument: ${arg}` }
|
|
2429
|
+
}
|
|
2430
|
+
}
|
|
2431
|
+
return { options }
|
|
2432
|
+
}
|
|
2433
|
+
|
|
2434
|
+
/**
|
|
2435
|
+
* `headlesscode pipeline` — run the stage-isolated research→filing pipeline
|
|
2436
|
+
* (issue #148) against one workspace. Research produces the artifact; filing
|
|
2437
|
+
* (unless --skip-filing) turns it into real GitHub issues.
|
|
2438
|
+
*/
|
|
2439
|
+
export async function pipelineMain(argv: string[]): Promise<number> {
|
|
2440
|
+
const { options, error } = parsePipelineArgs(argv)
|
|
2441
|
+
if (error) {
|
|
2442
|
+
process.stderr.write(`headlesscode pipeline: ${error}\n\n${PIPELINE_USAGE}`)
|
|
2443
|
+
return 2
|
|
2444
|
+
}
|
|
2445
|
+
if (options.help) {
|
|
2446
|
+
process.stdout.write(PIPELINE_USAGE)
|
|
2447
|
+
return 0
|
|
2448
|
+
}
|
|
2449
|
+
if (!options.workspace) {
|
|
2450
|
+
process.stderr.write(`headlesscode pipeline: --workspace <path> is required\n\n${PIPELINE_USAGE}`)
|
|
2451
|
+
return 2
|
|
2452
|
+
}
|
|
2453
|
+
|
|
2454
|
+
const workspaceRoot = path.resolve(options.workspace)
|
|
2455
|
+
process.stdout.write(`headlesscode pipeline: running research stage (mode researcher, artifact gate ON)...\n`)
|
|
2456
|
+
const research = await runResearchStage({
|
|
2457
|
+
workspaceRoot,
|
|
2458
|
+
taskText: options.researchTask,
|
|
2459
|
+
model: options.model,
|
|
2460
|
+
maxIterations: options.maxIterations,
|
|
2461
|
+
})
|
|
2462
|
+
if (research.status !== "ok" || !research.artifactPath) {
|
|
2463
|
+
process.stderr.write(
|
|
2464
|
+
`headlesscode pipeline: research stage failed — ${research.summary}\n`,
|
|
2465
|
+
)
|
|
2466
|
+
return 1
|
|
2467
|
+
}
|
|
2468
|
+
process.stdout.write(`headlesscode pipeline: research artifact at ${research.artifactPath}\n`)
|
|
2469
|
+
if (options.skipFiling) {
|
|
2470
|
+
return 0
|
|
2471
|
+
}
|
|
2472
|
+
|
|
2473
|
+
process.stdout.write(`headlesscode pipeline: running filing stage (mode issue-filer)...\n`)
|
|
2474
|
+
const filing = await runFilingStage({
|
|
2475
|
+
workspaceRoot,
|
|
2476
|
+
researchArtifactPath: research.artifactPath,
|
|
2477
|
+
model: options.model,
|
|
2478
|
+
maxIterations: options.maxIterations,
|
|
2479
|
+
})
|
|
2480
|
+
if (filing.status !== "ok") {
|
|
2481
|
+
process.stderr.write(
|
|
2482
|
+
`headlesscode pipeline: filing stage failed — ${filing.summary}\n`,
|
|
2483
|
+
)
|
|
2484
|
+
return 1
|
|
2485
|
+
}
|
|
2486
|
+
process.stdout.write(
|
|
2487
|
+
`headlesscode pipeline: filed issue(s): ${filing.issueNumbers.join(", ")}\n`,
|
|
2488
|
+
)
|
|
2489
|
+
return 0
|
|
2490
|
+
}
|
|
2491
|
+
|
|
2492
|
+
// ─── Main ────────────────────────────────────────────────────────────────────
|
|
2493
|
+
|
|
2494
|
+
export async function orchestrateMain(argv: string[]): Promise<number> {
|
|
2495
|
+
// Subcommand dispatch: `headlesscode orchestrate status ...` is the
|
|
2496
|
+
// read-side status/wait command, `... stop ...` stops a group's whole
|
|
2497
|
+
// worker tree (issue #20), `... cleanup ...` is the human-triggered
|
|
2498
|
+
// post-merge worktree cleanup, `... review/rework/resume ...` (issue #14)
|
|
2499
|
+
// are the standalone recovery subcommands; everything else is the
|
|
2500
|
+
// spawn+watch round.
|
|
2501
|
+
if (argv[0] === "status") {
|
|
2502
|
+
return statusMain(argv.slice(1))
|
|
2503
|
+
}
|
|
2504
|
+
if (argv[0] === "stop") {
|
|
2505
|
+
return stopMain(argv.slice(1))
|
|
2506
|
+
}
|
|
2507
|
+
if (argv[0] === "cleanup") {
|
|
2508
|
+
return cleanupMain(argv.slice(1))
|
|
2509
|
+
}
|
|
2510
|
+
if (argv[0] === "pipeline") {
|
|
2511
|
+
return pipelineMain(argv.slice(1))
|
|
2512
|
+
}
|
|
2513
|
+
// Issue #14 standalone recovery subcommands. Imported lazily (not
|
|
2514
|
+
// statically) because resume.ts imports this module's rework-decision
|
|
2515
|
+
// helpers — a static import here would create a module cycle.
|
|
2516
|
+
if (argv[0] === "review" || argv[0] === "rework" || argv[0] === "resume") {
|
|
2517
|
+
const { reviewMain, reworkMain, resumeMain } = await import("./resume.js")
|
|
2518
|
+
switch (argv[0]) {
|
|
2519
|
+
case "review":
|
|
2520
|
+
return reviewMain(argv.slice(1))
|
|
2521
|
+
case "rework":
|
|
2522
|
+
return reworkMain(argv.slice(1))
|
|
2523
|
+
default:
|
|
2524
|
+
return resumeMain(argv.slice(1))
|
|
2525
|
+
}
|
|
2526
|
+
}
|
|
2527
|
+
const { options, error } = parseOrchestrateArgs(argv)
|
|
2528
|
+
if (error) {
|
|
2529
|
+
process.stderr.write(`headlesscode orchestrate: ${error}\n\n${ORCHESTRATE_USAGE}`)
|
|
2530
|
+
return 2
|
|
2531
|
+
}
|
|
2532
|
+
if (options.help) {
|
|
2533
|
+
process.stdout.write(ORCHESTRATE_USAGE)
|
|
2534
|
+
return 0
|
|
2535
|
+
}
|
|
2536
|
+
if (!options.repo) {
|
|
2537
|
+
process.stderr.write(`headlesscode orchestrate: --repo <path> is required\n\n${ORCHESTRATE_USAGE}`)
|
|
2538
|
+
return 2
|
|
2539
|
+
}
|
|
2540
|
+
|
|
2541
|
+
const repo = path.resolve(options.repo)
|
|
2542
|
+
try {
|
|
2543
|
+
execFileSync("git", ["-C", repo, "rev-parse", "--git-dir"], { stdio: "ignore", timeout: 5000 })
|
|
2544
|
+
} catch {
|
|
2545
|
+
process.stderr.write(`headlesscode orchestrate: not a git repo: ${repo}\n`)
|
|
2546
|
+
return 2
|
|
2547
|
+
}
|
|
2548
|
+
|
|
2549
|
+
let issues: SplitIssue[]
|
|
2550
|
+
try {
|
|
2551
|
+
issues = loadIssues(options)
|
|
2552
|
+
} catch (err) {
|
|
2553
|
+
process.stderr.write(`headlesscode orchestrate: ${err instanceof Error ? err.message : String(err)}\n`)
|
|
2554
|
+
return 2
|
|
2555
|
+
}
|
|
2556
|
+
|
|
2557
|
+
// Fix 3: --file-issues (only valid with --issues-json) files a REAL GitHub
|
|
2558
|
+
// issue for every synthetic --issues-json entry and substitutes the real
|
|
2559
|
+
// number, so worktree/branch naming, task files, and PR-closing comments
|
|
2560
|
+
// all use a number that actually exists on GitHub. Runs before anything
|
|
2561
|
+
// else touches `issues`. A real, visible-to-GitHub side effect — never
|
|
2562
|
+
// silent: one confirmation line per created issue.
|
|
2563
|
+
if (options.fileIssues) {
|
|
2564
|
+
const ownerRepo = originOwnerRepo(repo)
|
|
2565
|
+
if (!ownerRepo) {
|
|
2566
|
+
process.stderr.write(
|
|
2567
|
+
`headlesscode orchestrate: --file-issues needs a parseable GitHub 'origin' remote on ${repo} ` +
|
|
2568
|
+
`(got none) to know where to file the issues\n`,
|
|
2569
|
+
)
|
|
2570
|
+
return 2
|
|
2571
|
+
}
|
|
2572
|
+
try {
|
|
2573
|
+
const { issues: filed, created } = fileSyntheticIssues(issues, (issue) => createGhIssue(repo, ownerRepo, issue))
|
|
2574
|
+
for (const c of created) {
|
|
2575
|
+
process.stdout.write(`[orchestrate] filed issue #${c.number}: ${c.title} -> ${c.url}\n`)
|
|
2576
|
+
}
|
|
2577
|
+
issues = filed
|
|
2578
|
+
} catch (err) {
|
|
2579
|
+
process.stderr.write(
|
|
2580
|
+
`headlesscode orchestrate: --file-issues failed: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
2581
|
+
)
|
|
2582
|
+
return 2
|
|
2583
|
+
}
|
|
2584
|
+
}
|
|
2585
|
+
|
|
2586
|
+
// Issue #53 pre-flight issue-size check + auto-split: a FREE deterministic
|
|
2587
|
+
// scan (split.ts's topLevelSectionCount) flags issues whose body reads
|
|
2588
|
+
// like 3+ independent pieces of work. Runs BEFORE splitIssues() below —
|
|
2589
|
+
// unlike the original warn-only version, this can REPLACE a flagged issue
|
|
2590
|
+
// with real sub-issues, and splitIssues must see the replacement, not the
|
|
2591
|
+
// oversized original. Auto-split needs a real (non-synthetic) issue to
|
|
2592
|
+
// close and a GitHub remote to file into; --issues-json entries and
|
|
2593
|
+
// remote-less repos fall back to warn-only (auto-split has nothing to
|
|
2594
|
+
// close/file against). --dry-run ALSO falls back to warn-only: filing
|
|
2595
|
+
// real issues and closing the parent are exactly the kind of visible,
|
|
2596
|
+
// hard-to-reverse side effects a dry run promises never to do.
|
|
2597
|
+
if (options.issueSizeCheck) {
|
|
2598
|
+
const warnings = issueSizeWarnings(issues)
|
|
2599
|
+
const ownerRepo = options.autoSplit && !options.issuesJson && !options.dryRun ? originOwnerRepo(repo) : undefined
|
|
2600
|
+
if (warnings.length > 0 && ownerRepo) {
|
|
2601
|
+
const model = options.model ?? DEFAULT_MODEL
|
|
2602
|
+
const llmClient = new OpenRouterClient({ apiKey: process.env.HEADLESSCODE_OPENROUTER_API_KEY })
|
|
2603
|
+
const { issues: split, outcomes } = await autoSplitOversizedIssues(issues, warnings, {
|
|
2604
|
+
proposeSplit: (issue) => proposeSemanticSplit(issue, llmClient, model),
|
|
2605
|
+
createIssue: (issue) => createGhIssue(repo, ownerRepo, issue),
|
|
2606
|
+
closeParent: (n, comment) => closeGhIssue(repo, ownerRepo, n, comment),
|
|
2607
|
+
})
|
|
2608
|
+
issues = split
|
|
2609
|
+
for (const o of outcomes) {
|
|
2610
|
+
if (o.outcome === "split") {
|
|
2611
|
+
process.stdout.write(
|
|
2612
|
+
`[orchestrate] auto-split #${o.number} ("${o.title}") into ${o.created?.length} sub-issue(s): ` +
|
|
2613
|
+
(o.created ?? []).map((c) => `#${c.number} (${c.url})`).join(", ") +
|
|
2614
|
+
` — parent closed\n`,
|
|
2615
|
+
)
|
|
2616
|
+
} else if (o.outcome === "kept-as-is") {
|
|
2617
|
+
process.stdout.write(
|
|
2618
|
+
`[orchestrate] #${o.number} ("${o.title}") flagged by the size heuristic, but the model determined ` +
|
|
2619
|
+
`it's one coherent piece of work — dispatching as-is\n`,
|
|
2620
|
+
)
|
|
2621
|
+
} else {
|
|
2622
|
+
process.stderr.write(
|
|
2623
|
+
`headlesscode orchestrate: WARNING: auto-split failed for #${o.number} ("${o.title}"): ${o.reason} — ` +
|
|
2624
|
+
`dispatching the original issue as-is. --no-auto-split silences future attempts.\n`,
|
|
2625
|
+
)
|
|
2626
|
+
}
|
|
2627
|
+
}
|
|
2628
|
+
} else if (warnings.length > 0) {
|
|
2629
|
+
printIssueSizeWarnings(warnings)
|
|
2630
|
+
}
|
|
2631
|
+
}
|
|
2632
|
+
|
|
2633
|
+
// Naming collision avoidance: a still-running round in the same repo
|
|
2634
|
+
// already occupies some `.worktrees/wN` dirs. Scan them and let
|
|
2635
|
+
// splitIssues name this round's groups around the gap instead of always
|
|
2636
|
+
// starting at w1 and colliding — see split.ts's occupiedNames param. Two
|
|
2637
|
+
// concurrent `orchestrate` invocations against the same repo can now
|
|
2638
|
+
// share it instead of the second one hard-failing on a "stale" worktree
|
|
2639
|
+
// that was actually just in-flight, not stale.
|
|
2640
|
+
let occupiedNames: Set<string> = new Set()
|
|
2641
|
+
try {
|
|
2642
|
+
occupiedNames = new Set(
|
|
2643
|
+
fs
|
|
2644
|
+
.readdirSync(path.join(repo, ".worktrees"), { withFileTypes: true })
|
|
2645
|
+
.filter((d) => d.isDirectory())
|
|
2646
|
+
.map((d) => d.name),
|
|
2647
|
+
)
|
|
2648
|
+
} catch {
|
|
2649
|
+
// No .worktrees dir yet — nothing occupied.
|
|
2650
|
+
}
|
|
2651
|
+
const specs = splitIssues(issues, { occupiedNames })
|
|
2652
|
+
|
|
2653
|
+
// Pre-spawn collision check: defense in depth only now that naming skips
|
|
2654
|
+
// occupied slots above — this should not fire in the normal case. It
|
|
2655
|
+
// still catches races (a worktree appearing between the scan above and
|
|
2656
|
+
// spawn) and fails loudly before spawning anything or writing task
|
|
2657
|
+
// files/state, so a skipped-at-spawn group can never be silently dropped.
|
|
2658
|
+
//
|
|
2659
|
+
// Liveness-aware (a real production incident, not hypothetical): a
|
|
2660
|
+
// generic "stale worktree" message previously read as an invitation to
|
|
2661
|
+
// manually `rm -rf`/`git worktree remove` the path — which, when the
|
|
2662
|
+
// worktree actually belonged to a still-running worker, destroyed live
|
|
2663
|
+
// session state mid-run and wasted real API spend, repeatedly, across
|
|
2664
|
+
// one dispatch session. Distinguish "still running" from "leftover" via
|
|
2665
|
+
// the SAME `.harness.pid` liveness check the watcher's own stall guard
|
|
2666
|
+
// uses (isPidAlive), and steer toward `orchestrate cleanup --apply`
|
|
2667
|
+
// specifically — NOT raw `git worktree remove`, which refuses on any
|
|
2668
|
+
// untracked file and can leave a stray `.headlesscode/` behind that
|
|
2669
|
+
// re-trips this exact check on the next attempt; `cleanup --apply`
|
|
2670
|
+
// removes harness artifacts (including `.headlesscode/`) and known-safe
|
|
2671
|
+
// residue BEFORE calling `git worktree remove`, precisely to avoid that
|
|
2672
|
+
// residue (see cleanup.ts's removeHarnessArtifacts/removeKnownSafeArtifacts).
|
|
2673
|
+
const staleWorktrees = specs
|
|
2674
|
+
.map((spec) => path.join(repo, ".worktrees", spec.name))
|
|
2675
|
+
.filter((wtPath) => fs.existsSync(wtPath))
|
|
2676
|
+
if (staleWorktrees.length > 0) {
|
|
2677
|
+
const details = staleWorktrees
|
|
2678
|
+
.map((wtPath) => {
|
|
2679
|
+
const rel = path.relative(repo, wtPath)
|
|
2680
|
+
return isPidAlive(wtPath) ? `${rel} (round IN PROGRESS — live pid)` : `${rel} (no live pid — likely leftover)`
|
|
2681
|
+
})
|
|
2682
|
+
.join(", ")
|
|
2683
|
+
process.stderr.write(
|
|
2684
|
+
`headlesscode orchestrate: worktree collision found: ${details}. ` +
|
|
2685
|
+
`If any are marked "round IN PROGRESS", do NOT delete them — that destroys a live session's state and spend. ` +
|
|
2686
|
+
`Retry once it finishes, or investigate with "headlesscode orchestrate status --repo ${repo}". ` +
|
|
2687
|
+
`For genuinely leftover worktrees, use "headlesscode orchestrate cleanup --repo ${repo} --apply" ` +
|
|
2688
|
+
`(only removes worktrees whose branch is verified merged, and clears .headlesscode/ before ` +
|
|
2689
|
+
`"git worktree remove" so it can't leave residue behind) — avoid raw "git worktree remove"/"rm -rf", ` +
|
|
2690
|
+
`which can leave a stray .headlesscode/ dir that re-trips this same check next time.\n`,
|
|
2691
|
+
)
|
|
2692
|
+
return 1
|
|
2693
|
+
}
|
|
2694
|
+
|
|
2695
|
+
if (options.dryRun) {
|
|
2696
|
+
// Per-mode model assignment: resolve each role's model exactly like a
|
|
2697
|
+
// real round so the dry-run preview shows the effective model per role
|
|
2698
|
+
// (mode-models.json / _default / OPENROUTER_MODEL; explicit --model
|
|
2699
|
+
// always wins). undefined -> the OpenRouter client's built-in default.
|
|
2700
|
+
const previewModel = (mode: string): string =>
|
|
2701
|
+
resolveModelForMode({ workspaceRoot: repo, mode, explicitModel: options.model, env: process.env }) ??
|
|
2702
|
+
"deepseek/deepseek-v4-flash-0731 (client default)"
|
|
2703
|
+
// Issue #16: estimate the round's expected cost/iterations from
|
|
2704
|
+
// recorded cost history, keyed by issue shape. Advisory only — a
|
|
2705
|
+
// history read failure must never fail the dry-run, just omit the
|
|
2706
|
+
// section.
|
|
2707
|
+
let historyRecords: Awaited<ReturnType<typeof readCostHistory>> = []
|
|
2708
|
+
try {
|
|
2709
|
+
historyRecords = await readCostHistory(repo)
|
|
2710
|
+
} catch {
|
|
2711
|
+
historyRecords = []
|
|
2712
|
+
}
|
|
2713
|
+
const estimateSection = buildEstimateSection(estimateGroups(historyRecords, specs, issues), historyRecords.length)
|
|
2714
|
+
printPlan(repo, options.mode, specs, issues, {
|
|
2715
|
+
worker: previewModel(options.mode),
|
|
2716
|
+
reviewer: previewModel(options.reviewMode),
|
|
2717
|
+
qa: previewModel(options.qaMode),
|
|
2718
|
+
}, options.maxIterations, estimateSection,
|
|
2719
|
+
options.planFirst
|
|
2720
|
+
? { mode: options.planFirstMode, maxIterations: options.planFirstMaxIterations }
|
|
2721
|
+
: undefined,
|
|
2722
|
+
)
|
|
2723
|
+
return 0
|
|
2724
|
+
}
|
|
2725
|
+
|
|
2726
|
+
if (!process.env.HEADLESSCODE_OPENROUTER_API_KEY) {
|
|
2727
|
+
process.stderr.write(
|
|
2728
|
+
"headlesscode orchestrate: HEADLESSCODE_OPENROUTER_API_KEY is not set (workers + reviewer need it).\n" +
|
|
2729
|
+
" Use --dry-run to preview the plan without an API key.\n",
|
|
2730
|
+
)
|
|
2731
|
+
return 2
|
|
2732
|
+
}
|
|
2733
|
+
|
|
2734
|
+
// 1a. Phase 6 GLOBAL concurrency cap (spec 6.3): abort BEFORE any work
|
|
2735
|
+
// (task files, spawn) when the fleet is already at/over the cap.
|
|
2736
|
+
// Cross-process view: orchestrator groups in spawned/running + watcher
|
|
2737
|
+
// in-flight 'spawned' entries. Fail-fast by design — orchestrate never
|
|
2738
|
+
// queues silently (the watcher's durable 'pending' state is the queueing
|
|
2739
|
+
// mechanism; see docs/phase6-cloud.md).
|
|
2740
|
+
const statePath = path.join(repo, ".worktrees", ".orchestrator-state.json")
|
|
2741
|
+
const activeNow = activeSessionCountForRepo(repo)
|
|
2742
|
+
if (activeNow >= options.maxConcurrentSessions) {
|
|
2743
|
+
process.stderr.write(
|
|
2744
|
+
`headlesscode orchestrate: concurrency cap reached — ${activeNow} active session(s) ` +
|
|
2745
|
+
`(cap ${options.maxConcurrentSessions}, set via --max-concurrent-sessions or ` +
|
|
2746
|
+
`HEADLESSCODE_MAX_CONCURRENT_SESSIONS). This round was NOT spawned. ` +
|
|
2747
|
+
`Wait for running sessions to finish or raise the cap.\n`,
|
|
2748
|
+
)
|
|
2749
|
+
return 1
|
|
2750
|
+
}
|
|
2751
|
+
|
|
2752
|
+
// 1a-sync. Issue #25: local master silently drifts unpushed from origin,
|
|
2753
|
+
// and every PR-merge cycle pays for it with re-merge conflict cascades.
|
|
2754
|
+
// At this "a new round is about to start" checkpoint, reconcile the two
|
|
2755
|
+
// (fetch origin, push local-only commits, fast-forward local onto origin)
|
|
2756
|
+
// so this round's branches start from what's actually on GitHub. Best-
|
|
2757
|
+
// effort auxiliary step: a failure is a loud warning, never an abort.
|
|
2758
|
+
// HEADLESSCODE_ORCHESTRATE_NO_SYNC=1 disables the mutation and only warns
|
|
2759
|
+
// on non-trivial drift (see git-sync.ts).
|
|
2760
|
+
if (process.env[ORCHESTRATE_SYNC_DISABLED_ENV]) {
|
|
2761
|
+
const drift = branchSyncStatus(repo)
|
|
2762
|
+
if (drift && drift.ahead > TRIVIAL_DRIFT_AHEAD) {
|
|
2763
|
+
process.stderr.write(
|
|
2764
|
+
`headlesscode orchestrate: WARNING: local ${drift.branch} is ${drift.ahead} commit(s) ahead of ${drift.remoteRef} ` +
|
|
2765
|
+
`(${ORCHESTRATE_SYNC_DISABLED_ENV} set — not pushing). Push it now to avoid merge-conflict cascades at ` +
|
|
2766
|
+
`PR-merge time (issue #25): git -C ${repo} push origin ${drift.branch}\n`,
|
|
2767
|
+
)
|
|
2768
|
+
}
|
|
2769
|
+
} else {
|
|
2770
|
+
const sync = syncBranchWithOrigin(repo)
|
|
2771
|
+
for (const line of syncSummaryLines(sync)) {
|
|
2772
|
+
process.stdout.write(`[sync] ${line}\n`)
|
|
2773
|
+
}
|
|
2774
|
+
for (const line of syncWarningLines(sync, repo)) {
|
|
2775
|
+
process.stderr.write(`headlesscode orchestrate: ${line}\n`)
|
|
2776
|
+
}
|
|
2777
|
+
}
|
|
2778
|
+
|
|
2779
|
+
// 1a-pre. Issue #13 pre-spawn preflight probe: a cheap 1-token completion
|
|
2780
|
+
// using the EXACT model + provider pin a real worker will use, so a dead
|
|
2781
|
+
// key / exhausted account balance / down pinned provider fails NOW instead
|
|
2782
|
+
// of 30-80 iterations (10-15 min + real spend) into the round. Runs before
|
|
2783
|
+
// task files or spawn; --no-preflight skips it for CI/non-interactive
|
|
2784
|
+
// contexts that don't want the extra round-trip. The probe's model is
|
|
2785
|
+
// resolved exactly like the worker's first spawn (same resolveModelForMode
|
|
2786
|
+
// precedence as 1c below), so what we probe is what the worker will call.
|
|
2787
|
+
let preflightProbe: PreflightResult | LocalPreflightResult | undefined
|
|
2788
|
+
if (options.preflight) {
|
|
2789
|
+
// 2026-08-27: probe whichever backend the WORKER's mode will actually
|
|
2790
|
+
// use (see cli.ts's useLocalCodeBackend / reviewer.ts's runReview /
|
|
2791
|
+
// qa.ts's runQa — same gate, repeated here since orchestrate never
|
|
2792
|
+
// otherwise imports cli.ts). Before this, a round configured entirely
|
|
2793
|
+
// for the local daemon still probed OpenRouter/DeepSeek unconditionally
|
|
2794
|
+
// and aborted on a cloud key that round would never touch.
|
|
2795
|
+
const useLocalBackendForWorker =
|
|
2796
|
+
process.env.HEADLESSCODE_CODE_MODE_BACKEND === "ollama" &&
|
|
2797
|
+
(process.env.HEADLESSCODE_LOCAL_BACKEND_MODES ?? "code")
|
|
2798
|
+
.split(",")
|
|
2799
|
+
.map((s) => s.trim())
|
|
2800
|
+
.filter(Boolean)
|
|
2801
|
+
.includes(options.mode)
|
|
2802
|
+
try {
|
|
2803
|
+
if (useLocalBackendForWorker) {
|
|
2804
|
+
preflightProbe = await runLocalPreflight({
|
|
2805
|
+
model: options.model ?? resolvePerModeEnv("HEADLESSCODE_CODE_MODE_MODEL", options.mode),
|
|
2806
|
+
baseUrl: resolvePerModeEnv("HEADLESSCODE_OLLAMA_URL", options.mode),
|
|
2807
|
+
})
|
|
2808
|
+
} else {
|
|
2809
|
+
const workerModelForProbe = resolveModelForMode({
|
|
2810
|
+
workspaceRoot: repo,
|
|
2811
|
+
mode: options.mode,
|
|
2812
|
+
explicitModel: options.model,
|
|
2813
|
+
env: process.env,
|
|
2814
|
+
})
|
|
2815
|
+
const maxCostRaw = process.env.HEADLESSCODE_MAX_COST_USD
|
|
2816
|
+
const maxCostUsd = maxCostRaw !== undefined && maxCostRaw !== "" ? Number(maxCostRaw) : undefined
|
|
2817
|
+
preflightProbe = await runPreflight({
|
|
2818
|
+
model: workerModelForProbe,
|
|
2819
|
+
workerSessions: specs.length,
|
|
2820
|
+
maxCostUsd: maxCostUsd !== undefined && Number.isFinite(maxCostUsd) && maxCostUsd > 0 ? maxCostUsd : undefined,
|
|
2821
|
+
})
|
|
2822
|
+
}
|
|
2823
|
+
} catch (err) {
|
|
2824
|
+
// Neither probe throws by contract, but if either ever does the
|
|
2825
|
+
// round must not proceed on an unchecked gate.
|
|
2826
|
+
process.stderr.write(
|
|
2827
|
+
`headlesscode orchestrate: preflight probe crashed unexpectedly: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
2828
|
+
)
|
|
2829
|
+
return 1
|
|
2830
|
+
}
|
|
2831
|
+
process.stdout.write(`[preflight] ${preflightProbe.line}\n`)
|
|
2832
|
+
if (preflightProbe.status !== "ok") {
|
|
2833
|
+
process.stderr.write(
|
|
2834
|
+
`headlesscode orchestrate: preflight FAILED (${preflightProbe.status}) — aborting before any workers spawn. ` +
|
|
2835
|
+
`Fix the problem above, or use --no-preflight to skip this check.\n`,
|
|
2836
|
+
)
|
|
2837
|
+
return 1
|
|
2838
|
+
}
|
|
2839
|
+
}
|
|
2840
|
+
|
|
2841
|
+
// 1b. Task files.
|
|
2842
|
+
const written = writeTaskFiles(repo, specs, issues, { planFirst: options.planFirst })
|
|
2843
|
+
process.stdout.write(`Wrote ${written.length} task file(s) under ${path.join(repo, "plans", "parallel-tasks")}\n`)
|
|
2844
|
+
|
|
2845
|
+
// 1c. Per-mode model assignment for the WORKER'S FIRST SPAWN. This must
|
|
2846
|
+
// resolve BEFORE step 2's spawnSync: the spawner script (and through it
|
|
2847
|
+
// run-worker.sh) inherits OPENROUTER_MODEL from the environment, so a
|
|
2848
|
+
// mode-models.json entry for the worker's mode would otherwise be silently
|
|
2849
|
+
// ignored on the first attempt (it only kicked in on rework re-spawns).
|
|
2850
|
+
// resolveModelForMode's own precedence holds here: an explicit --model
|
|
2851
|
+
// flag beats the config file, which beats OPENROUTER_MODEL.
|
|
2852
|
+
const { env: spawnEnv, workerModel } = buildSpawnEnv({
|
|
2853
|
+
repo,
|
|
2854
|
+
mode: options.mode,
|
|
2855
|
+
explicitModel: options.model,
|
|
2856
|
+
memoryDir: options.memoryDir,
|
|
2857
|
+
maxIterations: options.maxIterations,
|
|
2858
|
+
planFirst: options.planFirst,
|
|
2859
|
+
planFirstMode: options.planFirstMode,
|
|
2860
|
+
planFirstMaxIterations: options.planFirstMaxIterations,
|
|
2861
|
+
env: process.env,
|
|
2862
|
+
})
|
|
2863
|
+
|
|
2864
|
+
// 2. Spawn (the bash script is the actual spawner; cwd = target repo so
|
|
2865
|
+
// `git rev-parse --show-toplevel` resolves there).
|
|
2866
|
+
const spawnCmd = `bash ${path.join(HARNESS_ROOT_TS, "scripts", "spawn-parallel-worktrees.sh")} ${specs
|
|
2867
|
+
.map((spec) => `${spec.name}:${specs.indexOf(spec)}:plans/parallel-tasks/${spec.taskFile}`)
|
|
2868
|
+
.join(" ")}`
|
|
2869
|
+
process.stdout.write(`Spawn: ${spawnCmd}\n`)
|
|
2870
|
+
// `bash -c <spawnCmd>` (NOT spawnSync("bash", [spawnCmd], { shell: true }),
|
|
2871
|
+
// which would run `bash bash <script> …` and die with "cannot execute
|
|
2872
|
+
// binary file").
|
|
2873
|
+
const spawnResult = spawnSync("bash", ["-c", spawnCmd], {
|
|
2874
|
+
cwd: repo,
|
|
2875
|
+
env: spawnEnv,
|
|
2876
|
+
stdio: "inherit",
|
|
2877
|
+
})
|
|
2878
|
+
if (spawnResult.status !== 0) {
|
|
2879
|
+
process.stderr.write(`headlesscode orchestrate: spawn script failed (exit ${spawnResult.status})\n`)
|
|
2880
|
+
return 1
|
|
2881
|
+
}
|
|
2882
|
+
|
|
2883
|
+
// 3. Record batch + groups in the state file (the spawner already wrote
|
|
2884
|
+
// group entries; add the batch id).
|
|
2885
|
+
const batch = options.batch ?? `round-${new Date().toISOString().slice(0, 10)}`
|
|
2886
|
+
try {
|
|
2887
|
+
// Issue #80: the whole load-merge-save below runs inside mutateState's
|
|
2888
|
+
// cross-process lock. Without it, two concurrent `orchestrate` invocations
|
|
2889
|
+
// against the same repo (explicitly supported — see the concurrency
|
|
2890
|
+
// budget check above) can both loadState the same on-disk snapshot, then
|
|
2891
|
+
// whichever saveState runs last silently discards the other's round's
|
|
2892
|
+
// groups. patchGroup already guards single-group patches this way; this
|
|
2893
|
+
// is the same guarantee for this multi-group initial-write transaction.
|
|
2894
|
+
await mutateState(statePath, async (fresh) => {
|
|
2895
|
+
let state = fresh
|
|
2896
|
+
state.batch = batch
|
|
2897
|
+
// Issue #16: stamp every group with its per-issue shape (split.ts's
|
|
2898
|
+
// issueShape) at dispatch time — the only moment the issue titles/bodies
|
|
2899
|
+
// are in hand. cost-history recording happens LATER (when the group
|
|
2900
|
+
// reaches a terminal status, possibly in a separate orchestrate
|
|
2901
|
+
// invocation) and only has issue numbers; the shape is what lets
|
|
2902
|
+
// cost-estimate.ts match a NEW task to similar past ones, so it must be
|
|
2903
|
+
// captured here and persisted in the state file (OrchestratorGroup.shapes).
|
|
2904
|
+
// The titles/bodies themselves are captured at the same moment into
|
|
2905
|
+
// OrchestratorGroup.issueBodies: the rework/QA/continuation task files are
|
|
2906
|
+
// generated LATER from state alone (a separate orchestrate invocation) and
|
|
2907
|
+
// need them to embed each issue's real title/body inline instead of a
|
|
2908
|
+
// runtime `gh issue view` (which fails for synthetic --issues-json
|
|
2909
|
+
// numbers).
|
|
2910
|
+
const shapeByNumber = new Map<number, string>()
|
|
2911
|
+
const issueBodiesByNumber = new Map<number, { title: string; body?: string }>()
|
|
2912
|
+
for (const issue of issues) {
|
|
2913
|
+
shapeByNumber.set(issue.number, issueShape(issue))
|
|
2914
|
+
issueBodiesByNumber.set(issue.number, { title: issue.title, body: issue.body })
|
|
2915
|
+
}
|
|
2916
|
+
const groupNamesOnDisk = new Set(state.groups.map((g) => g.name))
|
|
2917
|
+
for (const spec of specs) {
|
|
2918
|
+
if (!groupNamesOnDisk.has(spec.name)) {
|
|
2919
|
+
continue
|
|
2920
|
+
}
|
|
2921
|
+
const shapes = spec.issues.map((n) => shapeByNumber.get(n) ?? "generic")
|
|
2922
|
+
const issueBodies: IssueBodies = {}
|
|
2923
|
+
for (const n of spec.issues) {
|
|
2924
|
+
const entry = issueBodiesByNumber.get(n)
|
|
2925
|
+
if (entry) {
|
|
2926
|
+
issueBodies[String(n)] = entry
|
|
2927
|
+
}
|
|
2928
|
+
}
|
|
2929
|
+
state = updateGroup(state, spec.name, { shapes, issueBodies })
|
|
2930
|
+
}
|
|
2931
|
+
// Issue #13: persist the preflight probe so the dashboard can surface the
|
|
2932
|
+
// preflight line on the round view (see aggregate.ts RoundSummary.preflight).
|
|
2933
|
+
if (preflightProbe) {
|
|
2934
|
+
// LocalPreflightResult has no cost fields at all (local inference
|
|
2935
|
+
// is genuinely free) — default to 0/undefined rather than widen
|
|
2936
|
+
// PreflightRecord's required probeCostUsd to optional for a case
|
|
2937
|
+
// that's always a real number either way.
|
|
2938
|
+
state.preflight = {
|
|
2939
|
+
status: preflightProbe.status,
|
|
2940
|
+
model: preflightProbe.model,
|
|
2941
|
+
baseUrl: preflightProbe.baseUrl,
|
|
2942
|
+
line: preflightProbe.line,
|
|
2943
|
+
latencyMs: preflightProbe.latencyMs,
|
|
2944
|
+
probeCostUsd: "probeCostUsd" in preflightProbe ? preflightProbe.probeCostUsd : 0,
|
|
2945
|
+
roundCostEstimateUsd: "roundCostEstimateUsd" in preflightProbe ? preflightProbe.roundCostEstimateUsd : undefined,
|
|
2946
|
+
}
|
|
2947
|
+
}
|
|
2948
|
+
return state
|
|
2949
|
+
})
|
|
2950
|
+
} catch (err) {
|
|
2951
|
+
if (!(err instanceof SyntaxError)) {
|
|
2952
|
+
// Not a corrupt-JSON case (e.g. the cross-process lock timed out) —
|
|
2953
|
+
// nothing to back up, just report and abort loudly.
|
|
2954
|
+
process.stderr.write(
|
|
2955
|
+
`headlesscode orchestrate: failed to record round state in ${statePath}: ` +
|
|
2956
|
+
`${err instanceof Error ? err.message : String(err)}\n`,
|
|
2957
|
+
)
|
|
2958
|
+
return 1
|
|
2959
|
+
}
|
|
2960
|
+
// Issue #77: loadState (inside mutateState) only throws SyntaxError for a
|
|
2961
|
+
// genuinely corrupt/truncated state file (ENOENT is already handled
|
|
2962
|
+
// inside loadState and returns a fresh default). Silently replacing it
|
|
2963
|
+
// here would destroy every group's status/review verdict with no
|
|
2964
|
+
// warning. Back the corrupt file up so it isn't lost, then refuse to
|
|
2965
|
+
// start rather than clobber it.
|
|
2966
|
+
const backupPath = `${statePath}.corrupt-${Date.now()}`
|
|
2967
|
+
try {
|
|
2968
|
+
await fs.promises.rename(statePath, backupPath)
|
|
2969
|
+
} catch {
|
|
2970
|
+
/* best-effort backup; fall through to the loud abort either way */
|
|
2971
|
+
}
|
|
2972
|
+
process.stderr.write(
|
|
2973
|
+
`headlesscode orchestrate: state file ${statePath} is corrupt and could not be parsed ` +
|
|
2974
|
+
`(${err.message}). Backed up to ${backupPath}. ` +
|
|
2975
|
+
`Refusing to start with an empty state — restore or repair the backup, then retry.\n`,
|
|
2976
|
+
)
|
|
2977
|
+
return 1
|
|
2978
|
+
}
|
|
2979
|
+
|
|
2980
|
+
// 4. Watch + review. Per-mode model assignment: each role resolves its OWN
|
|
2981
|
+
// model — workers use the group's mode (resolved in 1c, above — the SAME
|
|
2982
|
+
// const reused here for the rework loop), the reviewer uses reviewMode, QA
|
|
2983
|
+
// uses qaMode. An explicit --model flag beats every mode's config entry
|
|
2984
|
+
// (resolveModelForMode precedence rule 1), so a blanket override still
|
|
2985
|
+
// works exactly as before.
|
|
2986
|
+
const reviewerModel = resolveModelForMode({
|
|
2987
|
+
workspaceRoot: repo,
|
|
2988
|
+
mode: options.reviewMode,
|
|
2989
|
+
explicitModel: options.model,
|
|
2990
|
+
env: process.env,
|
|
2991
|
+
})
|
|
2992
|
+
const qaModel = resolveModelForMode({
|
|
2993
|
+
workspaceRoot: repo,
|
|
2994
|
+
mode: options.qaMode,
|
|
2995
|
+
explicitModel: options.model,
|
|
2996
|
+
env: process.env,
|
|
2997
|
+
})
|
|
2998
|
+
|
|
2999
|
+
process.stdout.write("Watching for worker completion (.harness.done markers)...\n")
|
|
3000
|
+
await watchGroups({
|
|
3001
|
+
repoRoot: repo,
|
|
3002
|
+
statePath,
|
|
3003
|
+
pollIntervalMs: options.pollIntervalMs,
|
|
3004
|
+
reviewEnabled: options.review,
|
|
3005
|
+
qaEnabled: options.qa,
|
|
3006
|
+
onGroupUpdate: async (group) => {
|
|
3007
|
+
// Auto-continue on iteration-exhaustion (issue #2 Part 2): a worker
|
|
3008
|
+
// that failed ONLY because it hit --max-iterations gets a fresh
|
|
3009
|
+
// session on the SAME worktree (the partial state is on disk), up
|
|
3010
|
+
// to --max-continuations, then the group is marked needs-human.
|
|
3011
|
+
// Runs regardless of --no-review (this is about worker failure, not
|
|
3012
|
+
// review routing) and never triggers for budget stops or real
|
|
3013
|
+
// errors — handleIterationExhaustion's empty patch keeps "failed".
|
|
3014
|
+
if (group.status === "failed") {
|
|
3015
|
+
const decision = handleIterationExhaustion(
|
|
3016
|
+
group,
|
|
3017
|
+
repo,
|
|
3018
|
+
options.maxContinuations,
|
|
3019
|
+
options.mode,
|
|
3020
|
+
workerModel,
|
|
3021
|
+
options.maxIterations,
|
|
3022
|
+
)
|
|
3023
|
+
if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
|
|
3024
|
+
fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
|
|
3025
|
+
fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
|
|
3026
|
+
process.stdout.write(
|
|
3027
|
+
`[orchestrate] ${group.name} continuation ${decision.newContinuationCount}: ` +
|
|
3028
|
+
`previous worker hit the iteration cap; wrote ${path.relative(repo, decision.taskFilePath)}; ` +
|
|
3029
|
+
`re-spawning worker on the same worktree...\n`,
|
|
3030
|
+
)
|
|
3031
|
+
const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
|
|
3032
|
+
cwd: repo,
|
|
3033
|
+
env: {
|
|
3034
|
+
...process.env,
|
|
3035
|
+
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3036
|
+
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3037
|
+
},
|
|
3038
|
+
stdio: "inherit",
|
|
3039
|
+
})
|
|
3040
|
+
if (spawnResult.status !== 0) {
|
|
3041
|
+
// The continuation worker could not be launched — the
|
|
3042
|
+
// group cannot continue itself, so surface it as needing
|
|
3043
|
+
// a human instead of leaving a dangling "running" group.
|
|
3044
|
+
process.stderr.write(
|
|
3045
|
+
`[orchestrate] continuation spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
|
|
3046
|
+
`marking group needs-human\n`,
|
|
3047
|
+
)
|
|
3048
|
+
// patchGroup reloads the state file fresh immediately
|
|
3049
|
+
// before merging (under a write lock) rather than basing
|
|
3050
|
+
// the merge on `currentState` — a stale snapshot that may
|
|
3051
|
+
// predate other groups' concurrent writes now that
|
|
3052
|
+
// review/QA for multiple groups can run in parallel
|
|
3053
|
+
// (issue #24).
|
|
3054
|
+
await patchGroup(statePath, group.name, {
|
|
3055
|
+
status: "needs-human",
|
|
3056
|
+
continuationCount: decision.newContinuationCount,
|
|
3057
|
+
last_activity: {
|
|
3058
|
+
note: `continuation spawn failed after ${decision.newContinuationCount} attempt(s); needs human`,
|
|
3059
|
+
},
|
|
3060
|
+
})
|
|
3061
|
+
} else {
|
|
3062
|
+
// Spawn succeeded: reset the group so the watcher re-polls
|
|
3063
|
+
// it as "running" → "done/failed" → re-continued exactly
|
|
3064
|
+
// like a first attempt.
|
|
3065
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3066
|
+
}
|
|
3067
|
+
return true
|
|
3068
|
+
}
|
|
3069
|
+
if (decision.patch.status === "needs-human") {
|
|
3070
|
+
// Continuation cap reached: terminal "needs-human" outcome.
|
|
3071
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3072
|
+
process.stderr.write(
|
|
3073
|
+
`[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newContinuationCount} continuation ` +
|
|
3074
|
+
`attempt(s) and still hits the iteration cap — a human must look at this worktree.\n`,
|
|
3075
|
+
)
|
|
3076
|
+
return true
|
|
3077
|
+
}
|
|
3078
|
+
// Real failure (not iteration exhaustion): stays failed.
|
|
3079
|
+
return
|
|
3080
|
+
}
|
|
3081
|
+
|
|
3082
|
+
// Deterministic post-hoc log analysis: runs once per group as soon
|
|
3083
|
+
// as it reaches "done", unconditionally (no --review/--qa gate, no
|
|
3084
|
+
// LLM cost) — the same tool-call/error/stall/repeated-command
|
|
3085
|
+
// findings a human would otherwise get by hand-reading harness.log
|
|
3086
|
+
// + events.jsonl. Never blocks review/QA below; a failure here is
|
|
3087
|
+
// logged and swallowed.
|
|
3088
|
+
if (group.status === "done" && group.log_analysis === undefined) {
|
|
3089
|
+
try {
|
|
3090
|
+
const analysis = await analyzeWorktreeSessions(groupWorktreePath(repo, group))
|
|
3091
|
+
const patch = {
|
|
3092
|
+
log_analysis: analysis
|
|
3093
|
+
? {
|
|
3094
|
+
findings: analysis.findings,
|
|
3095
|
+
toolCallCounts: analysis.toolCallCounts,
|
|
3096
|
+
toolErrorCounts: analysis.toolErrorCounts,
|
|
3097
|
+
analyzedAt: new Date().toISOString(),
|
|
3098
|
+
}
|
|
3099
|
+
: (false as const),
|
|
3100
|
+
}
|
|
3101
|
+
await patchGroup(statePath, group.name, patch)
|
|
3102
|
+
if (analysis && analysis.findings.length > 0) {
|
|
3103
|
+
process.stdout.write(
|
|
3104
|
+
`[orchestrate] ${group.name} log analysis: ${analysis.findings.length} finding(s):\n` +
|
|
3105
|
+
analysis.findings.map((f) => ` - ${f}\n`).join(""),
|
|
3106
|
+
)
|
|
3107
|
+
}
|
|
3108
|
+
} catch (err) {
|
|
3109
|
+
process.stderr.write(
|
|
3110
|
+
`[orchestrate] log analysis of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
3111
|
+
)
|
|
3112
|
+
}
|
|
3113
|
+
}
|
|
3114
|
+
|
|
3115
|
+
if (!options.review) {
|
|
3116
|
+
return
|
|
3117
|
+
}
|
|
3118
|
+
if (group.status !== "done" || group.review_verdict !== undefined) {
|
|
3119
|
+
return
|
|
3120
|
+
}
|
|
3121
|
+
process.stdout.write(`[orchestrate] reviewing ${group.name} (branch ${group.branch ?? "?"})...\n`)
|
|
3122
|
+
try {
|
|
3123
|
+
const reviewWtPath = groupWorktreePath(repo, group)
|
|
3124
|
+
// See hasRealWorktreeChanges's doc comment (Tier 1 of the
|
|
3125
|
+
// 2026-08-28 fabricated-review incident fix): a "clean" verdict
|
|
3126
|
+
// is structurally impossible with zero real changes, checked
|
|
3127
|
+
// directly against git — never asked of (or trusted from) the
|
|
3128
|
+
// review LLM itself. Skips the review session entirely rather
|
|
3129
|
+
// than spend a call that has nothing real to verify.
|
|
3130
|
+
if (!hasRealWorktreeChanges(reviewWtPath)) {
|
|
3131
|
+
await patchGroup(statePath, group.name, {
|
|
3132
|
+
review_verdict: "finding",
|
|
3133
|
+
pending_review_findings: [
|
|
3134
|
+
"No real changes found in the worktree (empty diff vs the tracked upstream, no uncommitted changes outside harness bookkeeping files) — there is nothing for a review to verify. Automatic fail: a \"clean\" verdict is structurally impossible with zero changes.",
|
|
3135
|
+
],
|
|
3136
|
+
reviewed_at: new Date().toISOString(),
|
|
3137
|
+
last_activity: { note: "review skipped: worktree has no real changes (automatic fail)" },
|
|
3138
|
+
})
|
|
3139
|
+
process.stdout.write(
|
|
3140
|
+
`[orchestrate] AUTOMATIC FAIL: ${group.name}'s worktree has no real changes — skipping the review session (a "clean" verdict is structurally impossible with an empty diff)\n`,
|
|
3141
|
+
)
|
|
3142
|
+
return true
|
|
3143
|
+
}
|
|
3144
|
+
const rawResult = await runReviewWithRetries({
|
|
3145
|
+
workspaceRoot: reviewWtPath,
|
|
3146
|
+
mode: options.reviewMode,
|
|
3147
|
+
model: reviewerModel,
|
|
3148
|
+
issues: group.issues,
|
|
3149
|
+
})
|
|
3150
|
+
// A review-SESSION failure must not feed into the rework-a-worker
|
|
3151
|
+
// path below — see handleReviewSessionError's doc comment.
|
|
3152
|
+
if (rawResult.verdict === "error") {
|
|
3153
|
+
await patchGroup(statePath, group.name, handleReviewSessionError(rawResult))
|
|
3154
|
+
process.stderr.write(
|
|
3155
|
+
`[orchestrate] NEEDS-HUMAN: ${group.name}'s review session failed repeatedly — ` +
|
|
3156
|
+
`${rawResult.summary}\n`,
|
|
3157
|
+
)
|
|
3158
|
+
return true
|
|
3159
|
+
}
|
|
3160
|
+
// See hasRealVerificationActivity's doc comment (Tier 2 of the
|
|
3161
|
+
// 2026-08-28 incident fix): a "clean" claim backed by zero real,
|
|
3162
|
+
// successful execute_command results in the session's OWN
|
|
3163
|
+
// transcript is downgraded to a finding rather than trusted —
|
|
3164
|
+
// the exact gap that let a fabricated review through even with
|
|
3165
|
+
// the required structured verdict line.
|
|
3166
|
+
let result = rawResult
|
|
3167
|
+
if (rawResult.verdict === "clean" && !hasRealVerificationActivity(reviewWtPath, rawResult.reportPath)) {
|
|
3168
|
+
process.stdout.write(
|
|
3169
|
+
`[orchestrate] note: ${group.name}'s review verdict was "clean" but the session's own transcript shows no real, successful execute_command result — downgrading to a finding rather than trust an unverified claim\n`,
|
|
3170
|
+
)
|
|
3171
|
+
result = {
|
|
3172
|
+
...rawResult,
|
|
3173
|
+
verdict: "finding",
|
|
3174
|
+
findings: [
|
|
3175
|
+
...rawResult.findings,
|
|
3176
|
+
"Review declared \"clean\" but its own transcript shows no real, successful execute_command result — nothing to substantiate the verdict was actually run. Treated as a finding rather than trusted.",
|
|
3177
|
+
],
|
|
3178
|
+
}
|
|
3179
|
+
}
|
|
3180
|
+
const patch = {
|
|
3181
|
+
review_verdict: result.verdict,
|
|
3182
|
+
pending_review_findings: result.findings,
|
|
3183
|
+
reviewed_at: new Date().toISOString(),
|
|
3184
|
+
// Issue #34: point at the review session's complete final
|
|
3185
|
+
// report so the full reasoning behind the verdict is one
|
|
3186
|
+
// file-read away, not a re-run away.
|
|
3187
|
+
review_report: result.reportPath,
|
|
3188
|
+
last_activity: {
|
|
3189
|
+
note: `reviewed: verdict=${result.verdict} (${result.findings.length} finding(s))`,
|
|
3190
|
+
},
|
|
3191
|
+
}
|
|
3192
|
+
await patchGroup(statePath, group.name, patch)
|
|
3193
|
+
process.stdout.write(
|
|
3194
|
+
`[orchestrate] ${group.name} review verdict: ${result.verdict} (${result.findings.length} finding(s))\n`,
|
|
3195
|
+
)
|
|
3196
|
+
|
|
3197
|
+
// Rework loop (plans/rework-loop.md): a "finding" verdict is
|
|
3198
|
+
// treated as NEW WORK — re-spawn a worker on the SAME worktree to
|
|
3199
|
+
// fix the findings, up to --max-rework-cycles attempts. Returning
|
|
3200
|
+
// true tells watchGroups to reload state from disk so it re-polls
|
|
3201
|
+
// the group as "running" → "done" → re-reviewed (the watcher's
|
|
3202
|
+
// in-memory state is otherwise stale — see watchGroups).
|
|
3203
|
+
if (result.verdict === "finding") {
|
|
3204
|
+
const decision = handleReviewVerdict(
|
|
3205
|
+
group,
|
|
3206
|
+
result,
|
|
3207
|
+
repo,
|
|
3208
|
+
options.maxReworkCycles,
|
|
3209
|
+
options.mode,
|
|
3210
|
+
workerModel,
|
|
3211
|
+
options.maxIterations,
|
|
3212
|
+
)
|
|
3213
|
+
if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
|
|
3214
|
+
fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
|
|
3215
|
+
fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
|
|
3216
|
+
process.stdout.write(
|
|
3217
|
+
`[orchestrate] ${group.name} rework cycle ${decision.newReworkCount}: ` +
|
|
3218
|
+
`wrote ${path.relative(repo, decision.taskFilePath)}; re-spawning worker on the same worktree...\n`,
|
|
3219
|
+
)
|
|
3220
|
+
const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
|
|
3221
|
+
cwd: repo,
|
|
3222
|
+
env: {
|
|
3223
|
+
...process.env,
|
|
3224
|
+
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3225
|
+
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3226
|
+
},
|
|
3227
|
+
stdio: "inherit",
|
|
3228
|
+
})
|
|
3229
|
+
if (spawnResult.status !== 0) {
|
|
3230
|
+
// The rework worker could not be launched — the group
|
|
3231
|
+
// cannot fix itself, so surface it as needing a human
|
|
3232
|
+
// instead of leaving a dangling "running" group.
|
|
3233
|
+
process.stderr.write(
|
|
3234
|
+
`[orchestrate] rework worker spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
|
|
3235
|
+
`marking group needs-human\n`,
|
|
3236
|
+
)
|
|
3237
|
+
await patchGroup(statePath, group.name, {
|
|
3238
|
+
status: "needs-human",
|
|
3239
|
+
reworkCount: decision.newReworkCount,
|
|
3240
|
+
last_activity: {
|
|
3241
|
+
note: `rework spawn failed after ${decision.newReworkCount} attempt(s); needs human`,
|
|
3242
|
+
},
|
|
3243
|
+
})
|
|
3244
|
+
} else {
|
|
3245
|
+
// Spawn succeeded: reset the group so the watcher
|
|
3246
|
+
// re-polls it as "running" → "done" → re-reviewed.
|
|
3247
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3248
|
+
}
|
|
3249
|
+
return true
|
|
3250
|
+
}
|
|
3251
|
+
// Cap reached: terminal "needs-human" outcome. Findings stay
|
|
3252
|
+
// recorded so a human can see exactly what the reviewer flagged.
|
|
3253
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3254
|
+
process.stderr.write(
|
|
3255
|
+
`[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newReworkCount} rework ` +
|
|
3256
|
+
`attempt(s) and the review still has findings — a human must look at this worktree.\n`,
|
|
3257
|
+
)
|
|
3258
|
+
return true
|
|
3259
|
+
}
|
|
3260
|
+
} catch (err) {
|
|
3261
|
+
process.stderr.write(
|
|
3262
|
+
`[orchestrate] review of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
3263
|
+
)
|
|
3264
|
+
}
|
|
3265
|
+
// Phase 4: after a group's workers complete AND review passes, run a
|
|
3266
|
+
// headless QA session against that worktree. QA only runs when review
|
|
3267
|
+
// was clean (or review was disabled); the result is recorded in the
|
|
3268
|
+
// group's `qa` field. QA uses read+command tools only — it can boot
|
|
3269
|
+
// the app and run tests but cannot modify source files.
|
|
3270
|
+
if (!options.qa) {
|
|
3271
|
+
return
|
|
3272
|
+
}
|
|
3273
|
+
if (group.status !== "done" || group.qa !== undefined) {
|
|
3274
|
+
return
|
|
3275
|
+
}
|
|
3276
|
+
// The callback's `group` is the pre-review snapshot; the review
|
|
3277
|
+
// verdict lives in the state file we just persisted via patchGroup,
|
|
3278
|
+
// so read it fresh from disk rather than any in-memory copy —
|
|
3279
|
+
// review/QA for other groups can be writing concurrently now
|
|
3280
|
+
// (issue #24), so a captured snapshot here could be stale.
|
|
3281
|
+
const updatedGroup = loadStateSync(statePath).groups.find((g) => g.name === group.name)
|
|
3282
|
+
const reviewPassed = !options.review || updatedGroup?.review_verdict === "clean"
|
|
3283
|
+
if (!reviewPassed) {
|
|
3284
|
+
return
|
|
3285
|
+
}
|
|
3286
|
+
process.stdout.write(`[orchestrate] QA ${group.name} (branch ${group.branch ?? "?"})...\n`)
|
|
3287
|
+
try {
|
|
3288
|
+
const qaWtPath = groupWorktreePath(repo, group)
|
|
3289
|
+
// See hasRealWorktreeChanges's doc comment: same automatic-fail
|
|
3290
|
+
// gate as the review step above, checked directly against git
|
|
3291
|
+
// before the QA LLM session ever runs — a "pass" verdict is
|
|
3292
|
+
// structurally impossible with zero real changes.
|
|
3293
|
+
if (!hasRealWorktreeChanges(qaWtPath)) {
|
|
3294
|
+
await patchGroup(statePath, group.name, {
|
|
3295
|
+
qa: {
|
|
3296
|
+
status: "failed",
|
|
3297
|
+
verdict: "fail",
|
|
3298
|
+
evidence:
|
|
3299
|
+
"No real changes found in the worktree (empty diff vs the tracked upstream, no uncommitted changes outside harness bookkeeping files) — there is nothing for QA to verify. Automatic fail: a \"pass\" verdict is structurally impossible with zero changes.",
|
|
3300
|
+
updated: new Date().toISOString(),
|
|
3301
|
+
},
|
|
3302
|
+
last_activity: { note: "QA skipped: worktree has no real changes (automatic fail)" },
|
|
3303
|
+
})
|
|
3304
|
+
process.stdout.write(
|
|
3305
|
+
`[orchestrate] AUTOMATIC FAIL: ${group.name}'s worktree has no real changes — skipping the QA session (a "pass" verdict is structurally impossible with an empty diff)\n`,
|
|
3306
|
+
)
|
|
3307
|
+
return true
|
|
3308
|
+
}
|
|
3309
|
+
const rawQaResult = await runQaWithRetries({
|
|
3310
|
+
workspaceRoot: qaWtPath,
|
|
3311
|
+
mode: options.qaMode,
|
|
3312
|
+
model: qaModel,
|
|
3313
|
+
})
|
|
3314
|
+
// A QA-SESSION failure (crash/budget/mistake-limit/no result —
|
|
3315
|
+
// see runQaWithRetries) survived every retry: verdict "error",
|
|
3316
|
+
// distinct from a real "fail". Must escalate to needs-human, NOT
|
|
3317
|
+
// silently record a "failed"-looking QA with empty evidence and
|
|
3318
|
+
// leave the group's top-level status "done" — that would settle
|
|
3319
|
+
// the round (cost recorded, nothing left to retry) while QA never
|
|
3320
|
+
// actually validated anything. Caught live 2026-08-05 on issue
|
|
3321
|
+
// #18's own round: the QA-side twin of #33's review bug.
|
|
3322
|
+
if (rawQaResult.verdict === "error") {
|
|
3323
|
+
await patchGroup(statePath, group.name, handleQaSessionError(rawQaResult))
|
|
3324
|
+
process.stderr.write(
|
|
3325
|
+
`[orchestrate] NEEDS-HUMAN: ${group.name}'s QA session failed repeatedly — ` +
|
|
3326
|
+
`${rawQaResult.summary}\n`,
|
|
3327
|
+
)
|
|
3328
|
+
return
|
|
3329
|
+
}
|
|
3330
|
+
// See hasRealVerificationActivity's doc comment (Tier 2 of the
|
|
3331
|
+
// 2026-08-28 incident fix): same downgrade as the review step
|
|
3332
|
+
// above — a "pass" claim backed by zero real, successful
|
|
3333
|
+
// execute_command results in the session's OWN transcript is
|
|
3334
|
+
// downgraded to fail (and routed through the same rework loop
|
|
3335
|
+
// as a real QA fail below) rather than trusted.
|
|
3336
|
+
let qaResult = rawQaResult
|
|
3337
|
+
if (rawQaResult.verdict === "pass" && !hasRealVerificationActivity(qaWtPath, rawQaResult.reportPath)) {
|
|
3338
|
+
process.stdout.write(
|
|
3339
|
+
`[orchestrate] note: ${group.name}'s QA verdict was "pass" but the session's own transcript shows no real, successful execute_command result — downgrading to fail rather than trust an unverified claim\n`,
|
|
3340
|
+
)
|
|
3341
|
+
qaResult = {
|
|
3342
|
+
...rawQaResult,
|
|
3343
|
+
verdict: "fail",
|
|
3344
|
+
evidence:
|
|
3345
|
+
`QA declared "pass" but its own transcript shows no real, successful execute_command result — ` +
|
|
3346
|
+
`nothing to substantiate the verdict was actually run. Treated as fail rather than trusted.\n\n` +
|
|
3347
|
+
`Original evidence: ${rawQaResult.evidence}`,
|
|
3348
|
+
}
|
|
3349
|
+
}
|
|
3350
|
+
// A real QA FAIL verdict is NEW WORK, not a settled "done"
|
|
3351
|
+
// (issue #52, caught live 2026-08-05): the group passed review
|
|
3352
|
+
// but QA caught real problems. Auto-spawn a rework cycle on the
|
|
3353
|
+
// SAME worktree exactly like a review finding, feeding the QA
|
|
3354
|
+
// evidence into the task file instead of review findings, up to
|
|
3355
|
+
// --max-rework-cycles. At the cap the group goes terminal
|
|
3356
|
+
// needs-human — it must never silently settle as "done" with
|
|
3357
|
+
// only the nested qa.status showing "failed".
|
|
3358
|
+
if (qaResult.verdict === "fail") {
|
|
3359
|
+
const decision = handleQaVerdict(
|
|
3360
|
+
group,
|
|
3361
|
+
qaResult,
|
|
3362
|
+
repo,
|
|
3363
|
+
options.maxReworkCycles,
|
|
3364
|
+
options.mode,
|
|
3365
|
+
workerModel,
|
|
3366
|
+
options.maxIterations,
|
|
3367
|
+
)
|
|
3368
|
+
if (decision.shouldSpawn && decision.taskFilePath && decision.taskContent && decision.spawnCommand) {
|
|
3369
|
+
fs.mkdirSync(path.dirname(decision.taskFilePath), { recursive: true })
|
|
3370
|
+
fs.writeFileSync(decision.taskFilePath, decision.taskContent, "utf-8")
|
|
3371
|
+
process.stdout.write(
|
|
3372
|
+
`[orchestrate] ${group.name} QA-fail rework cycle ${decision.newReworkCount}: ` +
|
|
3373
|
+
`wrote ${path.relative(repo, decision.taskFilePath)}; re-spawning worker on the same worktree...\n`,
|
|
3374
|
+
)
|
|
3375
|
+
const spawnResult = spawnSync("bash", ["-c", decision.spawnCommand], {
|
|
3376
|
+
cwd: repo,
|
|
3377
|
+
env: {
|
|
3378
|
+
...process.env,
|
|
3379
|
+
HEADLESSCODE_ROOT: HARNESS_ROOT_TS,
|
|
3380
|
+
...(options.memoryDir ? { HEADLESSCODE_MEMORY_DIR: path.resolve(options.memoryDir) } : {}),
|
|
3381
|
+
},
|
|
3382
|
+
stdio: "inherit",
|
|
3383
|
+
})
|
|
3384
|
+
if (spawnResult.status !== 0) {
|
|
3385
|
+
// The QA-fail rework worker could not be launched —
|
|
3386
|
+
// the group cannot fix itself, so surface it as
|
|
3387
|
+
// needing a human instead of leaving a dangling
|
|
3388
|
+
// "running" group (same as the review-finding path).
|
|
3389
|
+
process.stderr.write(
|
|
3390
|
+
`[orchestrate] QA-fail rework worker spawn for ${group.name} failed (exit ${spawnResult.status}); ` +
|
|
3391
|
+
`marking group needs-human\n`,
|
|
3392
|
+
)
|
|
3393
|
+
await patchGroup(statePath, group.name, {
|
|
3394
|
+
status: "needs-human",
|
|
3395
|
+
reworkCount: decision.newReworkCount,
|
|
3396
|
+
qa: {
|
|
3397
|
+
status: "failed",
|
|
3398
|
+
verdict: qaResult.verdict,
|
|
3399
|
+
evidence: qaResult.evidence.slice(0, 4000),
|
|
3400
|
+
report: qaResult.reportPath,
|
|
3401
|
+
updated: new Date().toISOString(),
|
|
3402
|
+
},
|
|
3403
|
+
last_activity: {
|
|
3404
|
+
note: `QA-fail rework spawn failed after ${decision.newReworkCount} attempt(s); needs human`,
|
|
3405
|
+
},
|
|
3406
|
+
})
|
|
3407
|
+
} else {
|
|
3408
|
+
// Spawn succeeded: reset the group so the watcher
|
|
3409
|
+
// re-polls it as "running" → "done" → re-reviewed
|
|
3410
|
+
// → re-QA'd.
|
|
3411
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3412
|
+
}
|
|
3413
|
+
return true
|
|
3414
|
+
}
|
|
3415
|
+
// Cap reached: terminal "needs-human" outcome. The QA
|
|
3416
|
+
// verdict/evidence stay recorded so a human can see
|
|
3417
|
+
// exactly what failed.
|
|
3418
|
+
await patchGroup(statePath, group.name, decision.patch)
|
|
3419
|
+
process.stderr.write(
|
|
3420
|
+
`[orchestrate] NEEDS-HUMAN: ${group.name} exhausted ${decision.newReworkCount} rework ` +
|
|
3421
|
+
`attempt(s) and QA still fails — a human must look at this worktree.\n`,
|
|
3422
|
+
)
|
|
3423
|
+
return true
|
|
3424
|
+
}
|
|
3425
|
+
await patchGroup(statePath, group.name, {
|
|
3426
|
+
qa: {
|
|
3427
|
+
status: "done",
|
|
3428
|
+
verdict: "pass",
|
|
3429
|
+
evidence: qaResult.evidence.slice(0, 4000),
|
|
3430
|
+
// Issue #34: the state file keeps the lightweight parsed
|
|
3431
|
+
// fields for quick scanning; point at the QA session's
|
|
3432
|
+
// COMPLETE final report so the full reasoning behind the
|
|
3433
|
+
// verdict is one file-read away, not a re-run away.
|
|
3434
|
+
report: qaResult.reportPath,
|
|
3435
|
+
updated: new Date().toISOString(),
|
|
3436
|
+
},
|
|
3437
|
+
last_activity: {
|
|
3438
|
+
note: "QA: verdict=pass",
|
|
3439
|
+
},
|
|
3440
|
+
})
|
|
3441
|
+
process.stdout.write(
|
|
3442
|
+
`[orchestrate] ${group.name} QA verdict: pass (status done)\n`,
|
|
3443
|
+
)
|
|
3444
|
+
} catch (err) {
|
|
3445
|
+
process.stderr.write(
|
|
3446
|
+
`[orchestrate] QA of ${group.name} failed: ${err instanceof Error ? err.message : String(err)}\n`,
|
|
3447
|
+
)
|
|
3448
|
+
}
|
|
3449
|
+
},
|
|
3450
|
+
})
|
|
3451
|
+
|
|
3452
|
+
// Reload the state from disk: onGroupUpdate persisted the review/QA
|
|
3453
|
+
// patches through the state FILE (the watcher's in-memory copy only tracks
|
|
3454
|
+
// worker completion), so the authoritative post-round state is on disk.
|
|
3455
|
+
const finalState = await loadState(statePath)
|
|
3456
|
+
|
|
3457
|
+
const terminal = finalState.groups.filter(
|
|
3458
|
+
(g) => g.status === "done" || g.status === "failed" || g.status === "needs-human",
|
|
3459
|
+
)
|
|
3460
|
+
process.stdout.write(
|
|
3461
|
+
`\nRound complete: ${terminal.length}/${finalState.groups.length} groups terminal. State: ${statePath}\n`,
|
|
3462
|
+
)
|
|
3463
|
+
const failed = finalState.groups.filter((g) => g.status === "failed")
|
|
3464
|
+
if (failed.length > 0) {
|
|
3465
|
+
process.stderr.write(
|
|
3466
|
+
`headlesscode orchestrate: ${failed.length} group(s) failed: ${failed.map((g) => g.name).join(", ")}\n`,
|
|
3467
|
+
)
|
|
3468
|
+
return 1
|
|
3469
|
+
}
|
|
3470
|
+
// Rework-exhausted groups are a DISTINCT, final "needs a human" outcome —
|
|
3471
|
+
// not a mid-review group and not a generic failure. Exit 1 like a failure
|
|
3472
|
+
// (the round did not fully succeed) but the message names the cause.
|
|
3473
|
+
const needsHuman = finalState.groups.filter((g) => g.status === "needs-human")
|
|
3474
|
+
if (needsHuman.length > 0) {
|
|
3475
|
+
process.stderr.write(
|
|
3476
|
+
`headlesscode orchestrate: ${needsHuman.length} group(s) need a human (rework/continuation ` +
|
|
3477
|
+
`attempts exhausted): ${needsHuman.map((g) => g.name).join(", ")}. ` +
|
|
3478
|
+
`See ${statePath} for their review_verdict/pending_review_findings and continuationCount.\n`,
|
|
3479
|
+
)
|
|
3480
|
+
return 1
|
|
3481
|
+
}
|
|
3482
|
+
|
|
3483
|
+
// Phase 4: human-approval deploy gate (--deploy). Runs ONLY when every
|
|
3484
|
+
// group is done, no group failed, every review is clean (when review is
|
|
3485
|
+
// enabled) and every QA verdict is pass (when --qa was passed). The gate
|
|
3486
|
+
// itself is a hard stop: scripts/deploy-gate.sh never invokes the repo's
|
|
3487
|
+
// deploy-production.sh without explicit human approval.
|
|
3488
|
+
if (options.deploy) {
|
|
3489
|
+
const gate = deployGateReady(finalState, options)
|
|
3490
|
+
if (!gate.ready) {
|
|
3491
|
+
process.stderr.write(
|
|
3492
|
+
`headlesscode orchestrate: deploy gate SKIPPED — ${gate.reason}. No deploy was attempted.\n`,
|
|
3493
|
+
)
|
|
3494
|
+
return 1
|
|
3495
|
+
}
|
|
3496
|
+
const batch = options.batch ?? `round-${new Date().toISOString().slice(0, 10)}`
|
|
3497
|
+
const gateCmd = [
|
|
3498
|
+
`bash ${path.join(HARNESS_ROOT_TS, "scripts", "deploy-gate.sh")}`,
|
|
3499
|
+
`"${repo}"`,
|
|
3500
|
+
`--batch "${batch}"`,
|
|
3501
|
+
options.deployArgs ? `--deploy-args "${options.deployArgs}"` : "",
|
|
3502
|
+
]
|
|
3503
|
+
.filter(Boolean)
|
|
3504
|
+
.join(" ")
|
|
3505
|
+
process.stdout.write(`[orchestrate] running deploy gate...\n`)
|
|
3506
|
+
// Same `bash -c` pattern as the spawn call above (no double-bash).
|
|
3507
|
+
const gateResult = spawnSync("bash", ["-c", gateCmd], {
|
|
3508
|
+
cwd: repo,
|
|
3509
|
+
env: { ...process.env },
|
|
3510
|
+
stdio: "inherit",
|
|
3511
|
+
})
|
|
3512
|
+
if (gateResult.status === 3) {
|
|
3513
|
+
process.stderr.write(
|
|
3514
|
+
"headlesscode orchestrate: deploy DENIED by the human-approval gate. deploy-production.sh was NOT run.\n",
|
|
3515
|
+
)
|
|
3516
|
+
return 1
|
|
3517
|
+
}
|
|
3518
|
+
if (gateResult.status !== 0) {
|
|
3519
|
+
process.stderr.write(
|
|
3520
|
+
`headlesscode orchestrate: deploy gate failed (exit ${gateResult.status}). No deploy was attempted.\n`,
|
|
3521
|
+
)
|
|
3522
|
+
return 1
|
|
3523
|
+
}
|
|
3524
|
+
process.stdout.write("[orchestrate] deploy approved by the gate and executed.\n")
|
|
3525
|
+
}
|
|
3526
|
+
return 0
|
|
3527
|
+
}
|
|
3528
|
+
|
|
3529
|
+
/**
|
|
3530
|
+
* Phase 4: decide whether the deploy gate may run after a round. Fail-closed:
|
|
3531
|
+
* returns { ready: false, reason } unless every group is done, none failed,
|
|
3532
|
+
* every review is clean (when review is enabled) and every QA verdict is pass
|
|
3533
|
+
* (when --qa was passed). A group that never got a QA record (e.g. review
|
|
3534
|
+
* found issues) blocks the deploy.
|
|
3535
|
+
*/
|
|
3536
|
+
function deployGateReady(
|
|
3537
|
+
state: OrchestratorState,
|
|
3538
|
+
options: OrchestrateOptions,
|
|
3539
|
+
): { ready: boolean; reason?: string } {
|
|
3540
|
+
if (state.groups.length === 0) {
|
|
3541
|
+
return { ready: false, reason: "no groups in state" }
|
|
3542
|
+
}
|
|
3543
|
+
const anyFailed = state.groups.some((g) => g.status === "failed")
|
|
3544
|
+
if (anyFailed) {
|
|
3545
|
+
return { ready: false, reason: "one or more groups failed" }
|
|
3546
|
+
}
|
|
3547
|
+
// Rework/continuation-exhausted groups: a distinct, final "needs a human"
|
|
3548
|
+
// outcome — the deploy must never proceed while one exists, and the reason
|
|
3549
|
+
// must say why (not a generic failure, not an in-flight review).
|
|
3550
|
+
const anyNeedsHuman = state.groups.some((g) => g.status === "needs-human")
|
|
3551
|
+
if (anyNeedsHuman) {
|
|
3552
|
+
return { ready: false, reason: "one or more groups need human attention (rework/continuation attempts exhausted)" }
|
|
3553
|
+
}
|
|
3554
|
+
const anyNotDone = state.groups.some((g) => g.status !== "done")
|
|
3555
|
+
if (anyNotDone) {
|
|
3556
|
+
return { ready: false, reason: "not all groups reached status done" }
|
|
3557
|
+
}
|
|
3558
|
+
if (options.review) {
|
|
3559
|
+
const anyUnreviewed = state.groups.some((g) => g.review_verdict !== "clean")
|
|
3560
|
+
if (anyUnreviewed) {
|
|
3561
|
+
return { ready: false, reason: "not all groups reviewed clean" }
|
|
3562
|
+
}
|
|
3563
|
+
}
|
|
3564
|
+
if (options.qa) {
|
|
3565
|
+
const anyQaUnpassed = state.groups.some((g) => g.qa?.verdict !== "pass")
|
|
3566
|
+
if (anyQaUnpassed) {
|
|
3567
|
+
return { ready: false, reason: "not all groups passed QA" }
|
|
3568
|
+
}
|
|
3569
|
+
}
|
|
3570
|
+
return { ready: true }
|
|
3571
|
+
}
|