headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,503 @@
1
+ /**
2
+ * Headless reviewer (Phase 2) — an adversarial fresh-context verification run.
3
+ *
4
+ * The reviewer is a SECOND harness session run against a worker's worktree /
5
+ * branch, with the review-mode checklist
6
+ * (`shared/prompts/review-mode-prompt.md`) used as the system prompt
7
+ * (`systemPromptOverride`). It reuses the reviewer's core principles
8
+ * verbatim in spirit:
9
+ *
10
+ * - fresh context: the reviewer did not write the code and must not give it
11
+ * benefit of the doubt — a report's output is a claim, not evidence,
12
+ * until reproduced;
13
+ * - non-edit discipline: the executor is READ-ONLY (no write_to_file at
14
+ * all), so the reviewer can inspect, re-run commands (tests, `gh`,
15
+ * `git diff`) and report — but can never fix anything itself;
16
+ * - verdict: clean, or findings (which the caller maps to "reopen issue").
17
+ *
18
+ * The verdict is parsed from the session's `attempt_completion` result (or
19
+ * its text-only answer) via `parseReviewResult`, which is exported for unit
20
+ * testing.
21
+ */
22
+
23
+ import * as fs from "node:fs/promises"
24
+ import * as fsSync from "node:fs"
25
+ import * as path from "node:path"
26
+ import { execFileSync } from "node:child_process"
27
+ import { fileURLToPath } from "node:url"
28
+
29
+ import { HeadlessSession } from "../engine/loop.js"
30
+ import { Logger } from "../engine/logger.js"
31
+ import { OpenRouterClient } from "../llm/openrouter.js"
32
+ import { OllamaClient } from "../llm/ollama.js"
33
+ import { envBoolean, resolvePerModeEnv } from "../cli.js"
34
+ import { createReadOnlyHeadlessExecutor } from "../tools/executor.js"
35
+ import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native-tools/index.js"
36
+ import { addCustomInstructions } from "../vendor/zoo-code/src/core/prompts/sections/custom-instructions.js"
37
+ import type { ChatTool, LlmClient, SessionResult } from "../engine/types.js"
38
+ import type { SessionBudget } from "../budget/budget.js"
39
+
40
+ /** Default location of the reviewer checklist, relative to the harness repo. */
41
+ export const DEFAULT_REVIEW_PROMPT_PATH = "shared/prompts/review-mode-prompt.md"
42
+
43
+ /** Tools the reviewer may call: read-only + command + completion/reporting. */
44
+ const REVIEW_TOOL_NAMES = new Set([
45
+ "read_file",
46
+ "list_files",
47
+ "execute_command",
48
+ "attempt_completion",
49
+ "ask_followup_question",
50
+ ])
51
+
52
+ /** The harness repo root (parent of src/orchestrator). */
53
+ export const HARNESS_ROOT = fileURLToPath(new URL("../..", import.meta.url))
54
+
55
+ export interface ReviewOptions {
56
+ /** The worktree to review (checked out on the worker's branch). */
57
+ workspaceRoot: string
58
+ /** Mode slug for the session (default: deepseek-reviewer). */
59
+ mode?: string
60
+ /** Model id (default: env OPENROUTER_MODEL / client default). */
61
+ model?: string
62
+ /** Path to the reviewer system prompt (default: harness shared/prompts/…). */
63
+ reviewPromptPath?: string
64
+ /** LLM client; inject a fake in tests (default: OpenRouterClient). */
65
+ llmClient?: LlmClient
66
+ /** Review task text (default: built from the workspace/branch). */
67
+ taskText?: string
68
+ /**
69
+ * Issue number(s) this group was actually assigned, when known (the
70
+ * orchestrator always has this in `group.issues`). When provided (and
71
+ * `taskText` is not explicitly overridden), the default task text names
72
+ * them directly instead of asking the reviewer to discover what was
73
+ * worked on via `gh issue list --state closed`. 2026-08-27: verified
74
+ * live — a review session given the generic "figure out which issues
75
+ * were closed" task, against a group whose issue had NOT actually been
76
+ * filed to GitHub (a synthetic `--issues-json` test group, closed
77
+ * locally but with no real issue thread to find), stalled into repeated
78
+ * empty replies ("I need to review the work... let me first understand
79
+ * what issues were closed") across all 3 retry attempts and correctly
80
+ * escalated to NEEDS-HUMAN — the fail-closed path worked, but a review
81
+ * that already knows the issue number shouldn't have to search for it
82
+ * at all. This closes that gap generally, not just for the synthetic
83
+ * case: even in normal use, `gh issue list --state closed` can miss
84
+ * for other reasons (API lag, pagination, label filtering) and there's
85
+ * no reason to re-derive data the caller already has.
86
+ */
87
+ issues?: number[]
88
+ /**
89
+ * Iteration ceiling — a backstop, not the primary guard (see `budget`
90
+ * below). Default is deliberately generous (200): a review that
91
+ * genuinely needs to re-run tests, re-derive baselines, and check
92
+ * several claims can legitimately need many tool calls, and a tight cap
93
+ * cutting it off mid-work is worse than a real runaway — it silently
94
+ * loses the review's actual findings (see `runReview`'s session-failure
95
+ * fallback below) rather than ending the session cleanly. A real
96
+ * incident: a review that had ALREADY reopened the issue and posted its
97
+ * findings comment (real GitHub side effects, correct) hit the OLD
98
+ * default of 40 right before calling `attempt_completion`, so the
99
+ * orchestrator only saw a synthetic "session error" instead of the
100
+ * real finding text.
101
+ */
102
+ maxIterations?: number
103
+ /**
104
+ * Cost/duration backstop (SessionBudget's maxCostUsd/maxDurationMs) —
105
+ * the preferred way to bound a review session: a true runaway gets
106
+ * caught by spend or wall-clock time, not by an arbitrary tool-call
107
+ * count that penalizes legitimate thorough work. Omit for no budget
108
+ * (the harness's own defaults apply, if any).
109
+ */
110
+ budget?: Pick<SessionBudget, "maxCostUsd" | "maxDurationMs">
111
+ }
112
+
113
+ export interface ReviewResult {
114
+ /** Individual findings extracted from the reviewer's final report. */
115
+ findings: string[]
116
+ /**
117
+ * clean = nothing wrong found; finding = reviewer reported a real problem
118
+ * with the WORKER's code; error = the review SESSION ITSELF failed
119
+ * (crashed, hit its own mistake/budget limit) before ever producing a
120
+ * real verdict — deliberately distinct from "finding" so the caller
121
+ * retries the REVIEW, not the worker's already-fine code (see
122
+ * runReviewWithRetries; issue caught live 2026-08-05 — a review session's
123
+ * own bounded-failure was being treated exactly like a code finding and
124
+ * triggered a pointless worker rework cycle).
125
+ */
126
+ verdict: "clean" | "finding" | "error"
127
+ /** The reviewer's full final summary. */
128
+ summary: string
129
+ /**
130
+ * Issue #34: absolute path to the review session's complete final report
131
+ * (`<workspaceRoot>/.headlesscode/reports/<sessionId>.md`), when the
132
+ * session succeeded and the report write succeeded. The orchestrator
133
+ * persists this so the full reasoning behind a review verdict is one
134
+ * file-read away, not a re-run away. Absent for a session error (no
135
+ * report was ever produced).
136
+ */
137
+ reportPath?: string
138
+ }
139
+
140
+ /** The reviewer-mode tool schemas (read-only + command, no write tools). */
141
+ export function reviewTools(): ChatTool[] {
142
+ return getNativeTools().filter(
143
+ (t) => t.type === "function" && REVIEW_TOOL_NAMES.has(t.function.name),
144
+ ) as unknown as ChatTool[]
145
+ }
146
+
147
+ function currentBranch(workspaceRoot: string): string {
148
+ try {
149
+ return execFileSync("git", ["-C", workspaceRoot, "branch", "--show-current"], {
150
+ encoding: "utf-8",
151
+ timeout: 5000,
152
+ }).trim()
153
+ } catch {
154
+ return "current branch"
155
+ }
156
+ }
157
+
158
+ function defaultTaskText(workspaceRoot: string, issues?: number[]): string {
159
+ const targetLine =
160
+ issues && issues.length > 0
161
+ ? `Review the work done in this workspace (branch: ${currentBranch(workspaceRoot)}) for ` +
162
+ `issue${issues.length > 1 ? "s" : ""} ${issues.map((n) => `#${n}`).join(", ")} — that is exactly ` +
163
+ `what this group was assigned, so start there directly (\`gh issue view ${issues[0]}\`, etc.) ` +
164
+ "rather than searching for it. If the issue number doesn't resolve on GitHub (e.g. a locally-" +
165
+ "tracked or synthetic task never filed as a real issue), that is not a review blocker — fall " +
166
+ "back to the worktree's own git log and diff against its base branch as the source of truth " +
167
+ "for what was actually done, and review that directly instead of stalling on the missing issue " +
168
+ "thread.\n\n"
169
+ : `Review the work done in this workspace (branch: ${currentBranch(workspaceRoot)}) ` +
170
+ "exactly as your operating procedure instructs. For each issue the worker closed, read the " +
171
+ "closing report, read the real diff, and re-run every checkable claim yourself.\n\n"
172
+ return (
173
+ targetLine +
174
+ "Any scratch " +
175
+ "you need (probe scripts, temp output captures) goes in `.headlesscode/scratch/` inside this " +
176
+ "workspace — NEVER write to `/tmp` or any other path outside the workspace. When you are " +
177
+ "done, call attempt_completion with a structured summary: a Findings section listing anything " +
178
+ "wrong with file:line evidence (omit or write 'none' if clean), and the final baseline numbers " +
179
+ "you personally confirmed.\n\n" +
180
+ "The VERY LAST LINE of your attempt_completion result must be exactly one of:\n" +
181
+ "VERDICT: CLEAN\n" +
182
+ "VERDICT: FINDING\n" +
183
+ "Nothing else on that line — no prose, no punctuation, no markdown formatting. This is the ONLY " +
184
+ "line the orchestrator parses to decide whether to trigger a rework cycle; everything else in " +
185
+ "your report is for a human reader. Get this exactly right even when the rest of your report " +
186
+ "discusses both clean and problematic findings — the verdict reflects the OVERALL outcome " +
187
+ "(FINDING if you reopened ANY issue, CLEAN only if none needed reopening)."
188
+ )
189
+ }
190
+
191
+ /**
192
+ * Run one headless review session against `workspaceRoot` using the reviewer
193
+ * checklist as the system prompt override and a read-only executor. Returns
194
+ * the parsed verdict/findings/summary. A failed session (LLM error, max
195
+ * iterations, bounded-failure mistake limit) is reported as `verdict:
196
+ * "error"` with a synthetic finding, so the orchestrator never mistakes an
197
+ * inconclusive review for a clean one — but also never mistakes it for a
198
+ * real code finding either (see runReviewWithRetries, which callers should
199
+ * generally use instead of calling this directly).
200
+ */
201
+ export async function runReview(options: ReviewOptions): Promise<ReviewResult> {
202
+ const {
203
+ workspaceRoot,
204
+ mode = "deepseek-reviewer",
205
+ model,
206
+ reviewPromptPath = `${HARNESS_ROOT}${DEFAULT_REVIEW_PROMPT_PATH}`,
207
+ llmClient,
208
+ taskText,
209
+ issues,
210
+ maxIterations = 200,
211
+ budget,
212
+ } = options
213
+
214
+ // 2026-08-27: verified live — a review session couldn't find the `curlee`
215
+ // compiler at all ("curlee runtime is not available in this environment"),
216
+ // fell back to eyeballing source code instead of running real checks, and
217
+ // separately had no idea `.roo/rules/rules.md` (this workspace's own
218
+ // language/build reference — e.g. joeos's documented curlee compiler
219
+ // location and Curlee syntax rules) existed. Root cause: `systemPromptOverride`
220
+ // is used VERBATIM by loop.ts (`this.config.systemPromptOverride ?? (await
221
+ // buildPrompt(...))`), which means review sessions skip `buildPrompt`
222
+ // entirely and, with it, `addCustomInstructions` — the exact mechanism
223
+ // that splices `.roo/rules/` into a WORKER session's prompt. Review
224
+ // sessions were structurally blind to project-specific guidance that
225
+ // worker sessions always get. Splicing it in here closes that gap the
226
+ // same way buildSystemPrompt already does for workers (see prompt.ts).
227
+ const reviewPromptText = await fs.readFile(reviewPromptPath, "utf-8")
228
+ const workspaceCustomInstructions = await addCustomInstructions("", "", workspaceRoot, mode, {})
229
+ const systemPromptOverride = workspaceCustomInstructions
230
+ ? `${reviewPromptText}\n\n${workspaceCustomInstructions}`
231
+ : reviewPromptText
232
+ // Issue #142 follow-up: orchestrate's review pass built its own
233
+ // OpenRouterClient unconditionally, so a local-backend setup (e.g. a
234
+ // review daemon on its own GPU) had no effect on it — only the
235
+ // single-session `--mode` CLI path respected HEADLESSCODE_LOCAL_BACKEND_MODES.
236
+ // Mirror cli.ts's gate here so `mode` (default "deepseek-reviewer")
237
+ // routes to the same local daemon a direct CLI invocation would.
238
+ const useLocalBackend =
239
+ process.env.HEADLESSCODE_CODE_MODE_BACKEND === "ollama" &&
240
+ (process.env.HEADLESSCODE_LOCAL_BACKEND_MODES ?? "code")
241
+ .split(",")
242
+ .map((s) => s.trim())
243
+ .filter(Boolean)
244
+ .includes(mode)
245
+ // 2026-08-27: cli.ts's own useLocalCodeBackend path learned this the hard
246
+ // way (see its effectiveModel doc comment) — every downstream consumer of
247
+ // `model` (session-start logs, cost/usage records, the `request.model`
248
+ // OllamaClient sends, which OllamaClient.resolveModel() prefers over its
249
+ // own defaultModel) must see the LOCAL model id when local backend is
250
+ // active, not the cloud one, or a local review session logs/tags itself
251
+ // as e.g. "deepseek/deepseek-v4-flash-0731" throughout even though it
252
+ // never touches OpenRouter.
253
+ const effectiveModel = useLocalBackend ? (resolvePerModeEnv("HEADLESSCODE_CODE_MODE_MODEL", mode) ?? model) : model
254
+ const client =
255
+ llmClient ??
256
+ (useLocalBackend
257
+ ? new OllamaClient({
258
+ baseUrl: resolvePerModeEnv("HEADLESSCODE_OLLAMA_URL", mode),
259
+ defaultModel: effectiveModel,
260
+ // OllamaClient has its own independent abort timer
261
+ // (ollama.ts's DEFAULT_OLLAMA_TIMEOUT_MS, 300s), separate
262
+ // from the llmTimeoutMs passed to HeadlessSession below —
263
+ // verified live 2026-08-28 (cli.ts's LOCAL_LLM_TIMEOUT_MS
264
+ // doc comment has the full story): raising the session-
265
+ // level timeout alone still left review sessions dying at
266
+ // exactly 300000ms because this constructor never heard
267
+ // about it.
268
+ timeoutMs: useLocalBackend ? 630_000 : undefined,
269
+ })
270
+ : new OpenRouterClient({ apiKey: process.env.HEADLESSCODE_OPENROUTER_API_KEY, defaultModel: model }))
271
+
272
+ // Mirror every log line to <worktree>/review.log — the same visibility
273
+ // `run-worker.sh` gives a worker via harness.log (a plain `tail -f`
274
+ // target), which review/QA sessions never had: they run IN-PROCESS
275
+ // inside orchestrate rather than as a spawned subprocess with redirected
276
+ // stdout, so their activity was only ever visible by hand-parsing the
277
+ // structured `.headlesscode/events/*.jsonl` feed. Raised directly
278
+ // 2026-08-05: "frustrating that i can't see the review logs the same way
279
+ // i can the harness logs... i don't like having to hunt them down."
280
+ // Append-only (matches harness.log's own convention across
281
+ // retries/rework re-reviews) with a run-separator line per session.
282
+ const logFilePath = path.join(workspaceRoot, "review.log")
283
+ fsSync.appendFileSync(logFilePath, `\n===== headlesscode review start: ${new Date().toISOString()} =====\n`, "utf-8")
284
+ const logger = new Logger({ level: "info", filePath: logFilePath })
285
+
286
+ const session = new HeadlessSession({
287
+ workspaceRoot,
288
+ mode,
289
+ model: effectiveModel,
290
+ taskText: taskText ?? defaultTaskText(workspaceRoot, issues),
291
+ maxIterations,
292
+ budget,
293
+ systemPromptOverride,
294
+ tools: reviewTools(),
295
+ executor: createReadOnlyHeadlessExecutor(workspaceRoot),
296
+ llmClient: client,
297
+ logger,
298
+ // Issue #144: mirror cli.ts's local-backend cost-tracking skip — a
299
+ // review session on a local daemon (e.g. the 2080 review-daemon) has
300
+ // no real dollar cost either.
301
+ trackCost: !useLocalBackend,
302
+ // 2026-08-27: verified live — a local Qwen3.5-9B review session
303
+ // fabricated a fake `<tool_call>...</tool_call>` text block as its
304
+ // very first reply, never ran a single real verification command, and
305
+ // the harness's bare-text pragmatic-success fallback (see cli.ts's
306
+ // requireExplicitCompletion doc comment) accepted that garbage as a
307
+ // normal `status: "success"` result. Because the session-level result
308
+ // looked like an ordinary success, `parseReviewResult`'s deliberate
309
+ // fail-open default ("no finding marker found -> clean", see its own
310
+ // doc comment) then silently recorded verdict "clean" for a worker
311
+ // whose `git diff --stat` was completely empty — the exact false
312
+ // completion this review stage exists to catch. cli.ts's worker path
313
+ // was already hardened against this for local sessions
314
+ // (requireExplicitCompletion); the review path never got the same
315
+ // treatment, even though it runs on the same daemon and hits the same
316
+ // failure mode. Without this, a garbage local-reviewer reply is
317
+ // silently indistinguishable from a real "verified clean" verdict —
318
+ // forcing it here means a bare/fabricated reply becomes a
319
+ // non-completing mistake (retried, then a genuine session error) so
320
+ // runReviewWithRetries's existing fail-closed handling actually
321
+ // triggers instead of being bypassed.
322
+ requireExplicitCompletion: useLocalBackend && !envBoolean("HEADLESSCODE_ALLOW_TEXT_ONLY_COMPLETION"),
323
+ // Same gap as cli.ts's LOCAL_LLM_TIMEOUT_MS (see its doc comment for
324
+ // the full story): this session construction never set llmTimeoutMs
325
+ // at all, so a review session on the local daemon always used the
326
+ // generic DEFAULT_LLM_TIMEOUT_MS (300s) — shorter than the shim's
327
+ // own deliberately-raised 600s upstream patience, so the harness
328
+ // gives up first on a genuinely slow (not hung) local call.
329
+ llmTimeoutMs: useLocalBackend ? 630_000 : undefined,
330
+ // Same gap as cli.ts's LOCAL_MAX_TOKENS (see its doc comment for the
331
+ // full incident): unset here, a review session on the local daemon
332
+ // falls back to loop.ts's DEFAULT_MAX_TOKENS (32768) — sized for a
333
+ // cloud reasoning model's 128K+ window, not this daemon's real
334
+ // 65,536-token total context, where one runaway generation can crash
335
+ // the whole session outright.
336
+ maxTokens: useLocalBackend ? 8192 : undefined,
337
+ })
338
+
339
+ const result: SessionResult = await session.run()
340
+ if (result.status !== "success" || result.result === undefined) {
341
+ const error = result.error ?? "unknown review session error"
342
+ return {
343
+ findings: [`[review session error] ${error}`],
344
+ verdict: "error",
345
+ summary: `Review session failed: ${error}`,
346
+ }
347
+ }
348
+ return { ...parseReviewResult(result.result), reportPath: result.reportPath }
349
+ }
350
+
351
+ /**
352
+ * Run a review, retrying ONLY when the review SESSION itself failed
353
+ * (verdict "error" — a crash, budget stop, or bounded-failure mistake limit
354
+ * inside the review session), up to `maxRetries` additional attempts. A real
355
+ * "clean" or "finding" verdict is returned immediately, first try — this
356
+ * only guards against the review session's own infrastructure hiccups, not
357
+ * against re-litigating a real finding.
358
+ *
359
+ * Why this exists: `runReview` reports a failed session as `verdict:
360
+ * "error"` rather than silently treating an inconclusive review as clean —
361
+ * correct fail-closed behavior. But the ORIGINAL caller-side handling
362
+ * treated ANY non-clean verdict (including "error") as a real code finding
363
+ * and triggered a full worker rework cycle to "fix" it — pointlessly
364
+ * respawning a worker against a placeholder error message with nothing
365
+ * actionable in it. Caught live 2026-08-05 running issue #17's own round.
366
+ * After `maxRetries` failed attempts, the caller gets the final "error"
367
+ * result back and should escalate to needs-human (a human should look at
368
+ * why review sessions keep failing) rather than reworking the worker.
369
+ */
370
+ export async function runReviewWithRetries(options: ReviewOptions, maxRetries = 2): Promise<ReviewResult> {
371
+ let last: ReviewResult = { findings: [], verdict: "error", summary: "no attempt made" }
372
+ for (let attempt = 0; attempt <= maxRetries; attempt++) {
373
+ last = await runReview(options)
374
+ if (last.verdict !== "error") {
375
+ return last
376
+ }
377
+ }
378
+ return last
379
+ }
380
+
381
+ /**
382
+ * Parse a reviewer's final report into { findings, verdict, summary }.
383
+ *
384
+ * Verdict heuristics (deterministic, tested):
385
+ * - "finding" if the report contains reopen/finding/problem markers, OR
386
+ * explicit "verdict: finding" — ALWAYS wins, even if the same report
387
+ * also contains clean-sounding language elsewhere (see below);
388
+ * - "clean" if it contains explicit clean markers ("review clean",
389
+ * "verified clean", "no findings", "everything checks out") AND no
390
+ * finding marker;
391
+ * - default "clean" when the report states neither (the reviewer only
392
+ * reopens for real problems, so an unmarked report is treated as clean).
393
+ *
394
+ * A finding marker ALWAYS overrides a clean marker, never the reverse. A
395
+ * real reviewer report reopening one issue while confirming several OTHER
396
+ * claims checked out cleanly is a completely normal report shape (e.g. a
397
+ * "### Verified clean" section listing what's fine, alongside a separate
398
+ * "### Reopened" section naming the real problem) — treating the ambient
399
+ * presence of "verified clean" text as grounds to override an explicit
400
+ * "reopened" elsewhere in the SAME report was a real production bug: it
401
+ * silently classified a genuinely reopened review as "clean", which would
402
+ * have skipped the entire rework mechanism this classification exists to
403
+ * trigger. Never weaken this back to "finding && !clean" — a report is
404
+ * either clean (nothing wrong) or it has findings; it cannot be both, and
405
+ * when in doubt (both markers present), a real finding must win.
406
+ *
407
+ * The finding-marker word list is deliberately narrow — NOT bare words
408
+ * like "failed", "wrong", "regression", "does not", or even bare
409
+ * "reopen(ed)" on its own. Three real production bugs came from
410
+ * progressively narrowing an over-broad list, each caught live in the same
411
+ * session:
412
+ * 1. A clean marker ("verified clean") anywhere overrode an explicit
413
+ * finding elsewhere in the same report (fixed: finding always wins).
414
+ * 2. Bare "failed"/"wrong"/"regression" matched ordinary baseline-
415
+ * reporting prose ("12 failed, 1187 passed" — pre-existing, unrelated
416
+ * to the change) with zero real problem present (fixed: dropped those
417
+ * words entirely, kept only `reopen(ed)`).
418
+ * 3. Bare "reopen(ed)" STILL wasn't safe: a genuinely clean report can
419
+ * explain that something does NOT need action using the word
420
+ * "reopen" itself — "(pre-existing / out of scope, no reopen)",
421
+ * "Staying closed. No reopening warranted." — with zero negation-
422
+ * detection, "reopen" appearing ANYWHERE, including inside a
423
+ * sentence explicitly saying it's NOT happening, still tripped the
424
+ * finding branch.
425
+ *
426
+ * The fix for #3 is not another negation lookbehind (that class of patch
427
+ * — `(?<!no\s)` — already failed once for "failed"; it only protects the
428
+ * EXACT phrase it names, never generalizes to different phrasing). The
429
+ * reliable signal instead is a STRUCTURAL one: `reopen(ed)` only counts
430
+ * when it is
431
+ * (a) a section heading on its own ("### Reopened"), or
432
+ * (b) explicitly tied to a specific issue number within the same
433
+ * sentence ("reopened #83" / "#83 ... reopened" / "issue #83
434
+ * REOPENED"),
435
+ * because the reviewer's own required report format always associates a
436
+ * real reopen with the specific issue number it applies to — an
437
+ * incidental "no reopen" aside never does. Do not go back to a bare
438
+ * `\breopen(ed)?\b` match, and do not add generic words back to this list
439
+ * without a specific report shape that needs them AND a test proving no
440
+ * false-positive on normal baseline/prose/aside language.
441
+ */
442
+ export function parseReviewResult(text: string): ReviewResult {
443
+ const summary = text.trim()
444
+
445
+ // A real reopen: either an explicit "### Reopened" heading, or
446
+ // "reopen(ed)" tied to a specific issue number within ~30 chars on
447
+ // either side, never crossing a sentence boundary (period/newline) —
448
+ // see the function docstring for why bare "reopen(ed)" isn't enough.
449
+ const REOPENED_HEADING_RE = /(?:^|\n)#{1,6}\s*reopened?\b/i
450
+ const REOPENED_WITH_ISSUE_RE = /#\d+\b[^.\n]{0,30}\breopen(?:ed)?\b|\breopen(?:ed)?\b[^.\n]{0,30}#\d+\b/i
451
+ const VERDICT_FINDING_RE = /\bverdict\b[^.\n]*\bfinding\b/i
452
+
453
+ const isRealFindingLine = (line: string): boolean =>
454
+ REOPENED_HEADING_RE.test(`\n${line}`) || REOPENED_WITH_ISSUE_RE.test(line) || VERDICT_FINDING_RE.test(line)
455
+
456
+ // Extract a "## Findings" / "Findings:" section if present.
457
+ const findings: string[] = []
458
+ const section = summary.match(
459
+ /(?:^|\n)(?:#{1,6}\s*)?findings?\s*:?\s*\n([\s\S]*?)(?=\n#{1,6}\s|\n\s*(?:verdict|summary)\b|\n\s*(?:PR|issues?):|\s*$)/i,
460
+ )
461
+ if (section?.[1]) {
462
+ for (const line of section[1].split("\n")) {
463
+ const item = line.replace(/^[-*\d.\s)\]]+\s*/, "").trim()
464
+ if (item && !/^(verdict|summary)/i.test(item)) {
465
+ findings.push(item)
466
+ }
467
+ }
468
+ }
469
+ // Fallback: any line that names a REAL reopen (tied to an issue number
470
+ // or its own heading) — not just the bare word "finding"/"reopen".
471
+ if (findings.length === 0) {
472
+ for (const line of summary.split("\n")) {
473
+ const item = line.replace(/^[-*\d.\s)\]]+\s*/, "").trim()
474
+ if (isRealFindingLine(item) && item.length > 8) {
475
+ findings.push(item)
476
+ }
477
+ }
478
+ }
479
+
480
+ // Primary path: an explicit, structured "VERDICT: CLEAN"/"VERDICT: FINDING"
481
+ // line (required by defaultTaskText) is authoritative — exact match, no
482
+ // heuristics, so no future report phrasing can ever false-positive it.
483
+ // This exists because free-form regex heuristics on prose proved to have
484
+ // no ceiling on false positives: three separate real incidents (see the
485
+ // function docstring) each needed a NEW fix for a NEW phrasing the
486
+ // previous fix didn't anticipate. A rigid required line has no such
487
+ // ceiling — it's either present and exact, or it's absent.
488
+ const structuredVerdict = summary.match(/^VERDICT:\s*(CLEAN|FINDING)\s*$/im)
489
+ if (structuredVerdict) {
490
+ const verdict: "clean" | "finding" = structuredVerdict[1]!.toUpperCase() === "FINDING" ? "finding" : "clean"
491
+ return { findings: [...new Set(findings)], verdict, summary }
492
+ }
493
+
494
+ // Fallback for a session that didn't emit the required line (an older
495
+ // prompt, a model that ignored the instruction, or a hand-written test
496
+ // fixture) — the heuristic below, kept exactly as hardened by the three
497
+ // past incidents, but no longer the primary path.
498
+ const hasFindingMarker =
499
+ REOPENED_HEADING_RE.test(summary) || REOPENED_WITH_ISSUE_RE.test(summary) || VERDICT_FINDING_RE.test(summary)
500
+
501
+ const verdict: "clean" | "finding" = hasFindingMarker ? "finding" : "clean"
502
+ return { findings: [...new Set(findings)], verdict, summary }
503
+ }