headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
package/src/qa/qa.ts ADDED
@@ -0,0 +1,487 @@
1
+ /**
2
+ * Headless QA (Phase 4) — run the target repo's `qa-agent` mode headlessly.
3
+ *
4
+ * Decision documented in docs/phase4-qa.md: the `qa-agent` mode's
5
+ * core loop (start the app → run `node qa/runner.js` → read `qa/last-report.json`
6
+ * → correlate with the server log → diagnose → fix → verify) is driven entirely
7
+ * by read + execute_command (+ edit for remediation) tools. The Playwright MCP
8
+ * server referenced in the mode's customInstructions is OPTIONAL enrichment
9
+ * ("you can also use Playwright MCP tools directly") — the harness cannot serve
10
+ * MCP (the vendored McpHub is a stub and MCP tools are excluded from
11
+ * `selectToolsForMode`), so browser interaction is done via the repo's own
12
+ * Playwright CLI (`qa/runner.js`, `npx playwright ...`) through
13
+ * `execute_command`. That is choice (b)-via-CLI from the Phase 4 spec.
14
+ *
15
+ * `runQa` therefore runs a second harness session against a worktree in
16
+ * `--mode qa-agent` (spliced from the target repo's `.roomodes` +
17
+ * `.roo/rules-qa-agent/` automatically by the vendored prompt builder — same
18
+ * mechanism as the reviewer). When the target repo has no `qa-agent` mode, a
19
+ * generic QA checklist (`GENERIC_QA_CHECKLIST`) is used as a
20
+ * `systemPromptOverride` so the session still gets a real QA operating
21
+ * procedure.
22
+ *
23
+ * Non-edit discipline: QA may RUN the app + tests (execute_command) but must
24
+ * NOT be able to modify source files unexpectedly. `write_to_file` is absent
25
+ * from BOTH the advertised tool list (`qaTools()`) and the executor
26
+ * (`createQaHeadlessExecutor`) — a model that calls it gets a clear
27
+ * "not implemented" error and the loop's consecutive-mistake bound will
28
+ * eventually trip.
29
+ *
30
+ * Verdict parsing mirrors the reviewer (`src/orchestrator/reviewer.ts`) but is
31
+ * FAIL-CLOSED: a QA report that neither explicitly passes nor explicitly fails
32
+ * is treated as `fail`, because QA gates a deploy — an inconclusive QA run must
33
+ * never let a deployment through.
34
+ */
35
+
36
+ import { execFileSync } from "node:child_process"
37
+ import * as fs from "node:fs"
38
+ import * as path from "node:path"
39
+
40
+ import { HeadlessSession } from "../engine/loop.js"
41
+ import { Logger } from "../engine/logger.js"
42
+ import { OllamaClient } from "../llm/ollama.js"
43
+ import { OpenRouterClient } from "../llm/openrouter.js"
44
+ import { envBoolean, resolvePerModeEnv } from "../cli.js"
45
+ import { appendBrowserActionTool, appendCodeIntelTools, loadCustomModes } from "../engine/prompt.js"
46
+ import { createQaHeadlessExecutor } from "../tools/executor.js"
47
+ import { getNativeTools } from "../vendor/zoo-code/src/core/prompts/tools/native-tools/index.js"
48
+ import { addCustomInstructions } from "../vendor/zoo-code/src/core/prompts/sections/custom-instructions.js"
49
+ import type { ChatTool, LlmClient, SessionResult } from "../engine/types.js"
50
+ import type { MemoryStore } from "../memory/types.js"
51
+ import type { SessionBudget } from "../budget/budget.js"
52
+
53
+ /** Default mode slug used for QA sessions (the qa-agent mode). */
54
+ export const DEFAULT_QA_MODE = "qa-agent"
55
+
56
+ /** Tools the QA agent may call: read + list + command + reporting. NO write. */
57
+ const QA_TOOL_NAMES = new Set([
58
+ "read_file",
59
+ "list_files",
60
+ "execute_command",
61
+ "attempt_completion",
62
+ "ask_followup_question",
63
+ ])
64
+
65
+ /**
66
+ * The QA-mode tool schemas (read-only + command, no write tools), plus
67
+ * browser_action and the four code-intelligence tools — QA can boot the app,
68
+ * visually verify it in a headless browser (the phase4 gap "live
69
+ * Playwright-MCP interactivity is not available headlessly" is partially
70
+ * closed for agent-driven checks; see src/tools/browser/tool.ts), and
71
+ * navigate source to substantiate findings (see src/codeintel/).
72
+ */
73
+ export function qaTools(): ChatTool[] {
74
+ const tools = getNativeTools().filter(
75
+ (t) => t.type === "function" && QA_TOOL_NAMES.has(t.function.name),
76
+ ) as unknown as ChatTool[]
77
+ return appendCodeIntelTools(appendBrowserActionTool(tools))
78
+ }
79
+
80
+ /**
81
+ * Generic QA checklist used when the target repo defines no `qa-agent` mode.
82
+ * The text deliberately opens with "You are a QA agent" so e2e mock
83
+ * scenarios can classify QA sessions deterministically (see
84
+ * scripts/e2e/mock-openrouter.mjs `qa-orchestrate`).
85
+ */
86
+ export const GENERIC_QA_CHECKLIST = [
87
+ "You are a QA agent for this project. Your job is to verify the work done in this workspace",
88
+ "and report evidence with REAL command output. You operate in a loop: boot → test → exercise →",
89
+ "report. You do not write or edit source files — QA is verification only; report problems,",
90
+ "do not fix them.",
91
+ "",
92
+ "## How to perform QA",
93
+ "",
94
+ "0. FIRST, before anything else: run `git status --porcelain` and",
95
+ " `git diff <upstream>...HEAD` to see exactly what changed in this worktree — including",
96
+ " UNTRACKED files (this pipeline's workers frequently do not commit; an untracked new",
97
+ " file IS the change under review, same as a committed diff would be). Verify against",
98
+ " THAT real change, not against whatever existing file happens to be topically similar",
99
+ " or easiest to find — reviewing/exercising the wrong file produces a verdict about code",
100
+ " that was never actually touched.",
101
+ "1. This checklist is generic (`npm test`/`pytest`/\"boot the application\") because it is a",
102
+ " fallback for when the repo defines no project-specific QA mode — it will not literally",
103
+ " apply to every repo (e.g. a kernel/systems project has no `npm start` to boot). Discover",
104
+ " the REAL build/verify/boot commands from the repo itself first — check the Makefile",
105
+ " (`make help`, or just read it), README, CI config, and `scripts/` — and use those,",
106
+ " not whatever this checklist's generic examples happen to name.",
107
+ "2. Boot the application or relevant service using the repo's own start script / command.",
108
+ " Run it as a background process if it would block; set a timeout on every wait loop and",
109
+ " every command. Any scratch you need (probe scripts, temp output captures) goes in",
110
+ " `<workspace>/.headlesscode/scratch/` — NEVER write to `/tmp` or any other path outside",
111
+ " the workspace.",
112
+ "3. Run the project's real test/verification suite and capture the REAL output — counts of",
113
+ " passed/failed tests (or the equivalent pass/fail signal for this repo's toolchain) are",
114
+ " the primary evidence.",
115
+ "4. Exercise the changed behavior identified in step 0: run the relevant command, script,",
116
+ " or browser check (e.g. a Playwright runner script the repo already has) and capture output.",
117
+ "5. If anything fails, report it — do not modify source files to make it pass.",
118
+ "",
119
+ "## Definition of done",
120
+ "",
121
+ "A task is complete when every check ran clean and the changed behavior works as expected.",
122
+ "Report a final summary with evidence: the real commands you ran and their output, test",
123
+ "counts, and any errors found.",
124
+ "",
125
+ "## How to finish",
126
+ "",
127
+ "Call attempt_completion with a structured summary:",
128
+ "- A verdict line: `QA PASS` (all checks clean) or `QA FAIL` (any error, failed test, or",
129
+ " behavior that does not work).",
130
+ "- An `## Evidence` section listing each command you ran and its real output.",
131
+ "- The final baseline numbers you personally confirmed.",
132
+ ].join("\n")
133
+
134
+ export interface RunQaOptions {
135
+ /** The worktree to run QA against (checked out on the worker's branch). */
136
+ workspaceRoot: string
137
+ /** The QA checklist task text (default: built from the mode/workspace). */
138
+ taskText?: string
139
+ /** Mode slug (default: `qa-agent` — the QA mode, auto-spliced). */
140
+ mode?: string
141
+ /** Model id (default: env OPENROUTER_MODEL / client default). */
142
+ model?: string
143
+ /** LLM client; inject a fake in tests (default: OpenRouterClient). */
144
+ llmClient?: LlmClient
145
+ /** OpenRouter base URL override (e2e points this at the mock server). */
146
+ baseUrl?: string
147
+ /**
148
+ * Loop iteration cap — a backstop, not the primary guard (see `budget`
149
+ * below). Default is deliberately generous (200): QA that boots the
150
+ * app, runs a real test suite, and exercises changed behavior can
151
+ * legitimately need many tool calls, and cutting it off mid-work
152
+ * silently loses the real evidence rather than ending cleanly (see
153
+ * `runQa`'s session-failure fallback below) — a review session hitting
154
+ * this exact problem at the old default of 40 was a real incident.
155
+ */
156
+ maxIterations?: number
157
+ /**
158
+ * Cost/duration backstop (SessionBudget's maxCostUsd/maxDurationMs) —
159
+ * the preferred way to bound a QA session: a true runaway gets caught
160
+ * by spend or wall-clock time, not by an arbitrary tool-call count that
161
+ * penalizes legitimate thorough work.
162
+ */
163
+ budget?: Pick<SessionBudget, "maxCostUsd" | "maxDurationMs">
164
+ /** Phase 3 memory store (optional; QA can reuse the project's memory). */
165
+ memory?: MemoryStore | null
166
+ /** Project scope for memory (default: basename of workspaceRoot). */
167
+ project?: string
168
+ }
169
+
170
+ export type QaVerdict = "pass" | "fail" | "error"
171
+
172
+ export interface QaResult {
173
+ /** pass = all QA checks clean; fail = problems found; error = session error. */
174
+ verdict: QaVerdict
175
+ /** The evidence section extracted from the report (real command output). */
176
+ evidence: string
177
+ /** The full final summary from the QA session. */
178
+ summary: string
179
+ /**
180
+ * Issue #34: absolute path to the QA session's complete final report
181
+ * (`<workspaceRoot>/.headlesscode/reports/<sessionId>.md`), when the
182
+ * session succeeded and the report write succeeded. The orchestrator
183
+ * persists this so the full reasoning behind a QA verdict is one
184
+ * file-read away, not a re-run away. Absent for a session error (no
185
+ * report was ever produced).
186
+ */
187
+ reportPath?: string
188
+ }
189
+
190
+ function currentBranch(workspaceRoot: string): string {
191
+ try {
192
+ return execFileSync("git", ["-C", workspaceRoot, "branch", "--show-current"], {
193
+ encoding: "utf-8",
194
+ timeout: 5000,
195
+ }).trim()
196
+ } catch {
197
+ return "current branch"
198
+ }
199
+ }
200
+
201
+ function defaultTaskText(workspaceRoot: string, mode: string): string {
202
+ return (
203
+ `Run the QA checklist defined in your operating instructions against this workspace ` +
204
+ `(branch: ${currentBranch(workspaceRoot)}, mode: ${mode}).\n\n` +
205
+ `FIRST: run \`git diff origin/master...HEAD --stat\` (or the equivalent against this repo's ` +
206
+ `actual base branch) to see EXACTLY which files changed. That diff is your scope — it tells ` +
207
+ `you which behavior to exercise. A real incident: without this instruction, a QA session spent ` +
208
+ `over a dozen iterations exploring an unrelated chat-UI input component and an unrelated ` +
209
+ `dependency-pinning question before ever looking at the actual diff, on a change that only ` +
210
+ `touched two shell scripts and a docs file.\n\n` +
211
+ `Boot the app and confirm it starts cleanly (a QUICK smoke test — no new JS/console errors on ` +
212
+ `load — not a deep exploration of unrelated areas), run the relevant test suite, then exercise ` +
213
+ `SPECIFICALLY the behavior the diff touched. If the diff doesn't touch the browser-facing app at ` +
214
+ `all (e.g. only scripts/docs/backend), the boot-and-smoke-test step still applies but do not go ` +
215
+ `looking for unrelated things to click through — there's nothing in scope for deeper UI ` +
216
+ `interaction. Report EVIDENCE with real command output. ` +
217
+ `When you are done, call attempt_completion with a structured summary: an Evidence section ` +
218
+ `with the real commands you ran and their output, and the final baseline numbers you ` +
219
+ `personally confirmed.\n\n` +
220
+ `The VERY LAST LINE of your attempt_completion result must be exactly one of:\n` +
221
+ `QA_VERDICT: PASS\n` +
222
+ `QA_VERDICT: FAIL\n` +
223
+ `Nothing else on that line — no prose, no punctuation, no markdown formatting. This is the ONLY ` +
224
+ `line the orchestrator parses to decide pass/fail; everything else in your report is for a ` +
225
+ `human reader.`
226
+ )
227
+ }
228
+
229
+ /**
230
+ * Run one headless QA session against `workspaceRoot` using the target repo's
231
+ * `qa-agent` mode (auto-spliced from .roomodes + .roo/rules-qa-agent/) or a
232
+ * generic checklist when the mode is absent. Returns the parsed verdict /
233
+ * evidence / summary. A failed session (LLM error, max iterations) is reported
234
+ * as `verdict: "error"` — the caller must never treat an inconclusive QA run
235
+ * as a pass.
236
+ */
237
+ export async function runQa(options: RunQaOptions): Promise<QaResult> {
238
+ const {
239
+ workspaceRoot,
240
+ mode = DEFAULT_QA_MODE,
241
+ model,
242
+ llmClient,
243
+ baseUrl,
244
+ maxIterations = 200,
245
+ budget,
246
+ memory = null,
247
+ project,
248
+ } = options
249
+
250
+ // Detect whether the target repo actually defines the requested mode. If
251
+ // not, fall back to the generic checklist as a system prompt override.
252
+ const customModes = await loadCustomModes(workspaceRoot)
253
+ const modeExists = customModes.some((m) => m.slug === mode)
254
+ // 2026-08-27 (mirrors reviewer.ts's runReview — same incident): when
255
+ // modeExists is true, systemPromptOverride stays undefined and loop.ts's
256
+ // own buildPrompt() path already splices in `.roo/rules/` via
257
+ // addCustomInstructions. But the GENERIC_QA_CHECKLIST fallback is used
258
+ // VERBATIM by loop.ts, bypassing buildPrompt (and addCustomInstructions
259
+ // with it) entirely — a QA session taking this fallback path would be
260
+ // just as blind to project-specific guidance (e.g. joeos's documented
261
+ // curlee compiler location) as the review session was found to be.
262
+ const workspaceCustomInstructions = modeExists ? "" : await addCustomInstructions("", "", workspaceRoot, mode, {})
263
+ const systemPromptOverride = modeExists
264
+ ? undefined
265
+ : workspaceCustomInstructions
266
+ ? `${GENERIC_QA_CHECKLIST}\n\n${workspaceCustomInstructions}`
267
+ : GENERIC_QA_CHECKLIST
268
+
269
+ // Mirror reviewer.ts's runReview gate (issue #142 follow-up): unlike
270
+ // reviewer.ts, this constructed an OpenRouterClient unconditionally, so
271
+ // a local-backend setup (HEADLESSCODE_CODE_MODE_BACKEND=ollama +
272
+ // HEADLESSCODE_LOCAL_BACKEND_MODES including this QA mode) had no effect
273
+ // on QA sessions at all, even though the worker and reviewer both
274
+ // respected it. Same gate, same precedence (an explicit llmClient/model/
275
+ // baseUrl injected by a caller still wins).
276
+ const useLocalBackend =
277
+ process.env.HEADLESSCODE_CODE_MODE_BACKEND === "ollama" &&
278
+ (process.env.HEADLESSCODE_LOCAL_BACKEND_MODES ?? "code")
279
+ .split(",")
280
+ .map((s) => s.trim())
281
+ .filter(Boolean)
282
+ .includes(mode)
283
+ // Same fix as reviewer.ts's effectiveModel (2026-08-27): downstream
284
+ // consumers of `model` (session-start logs, cost records, the
285
+ // `request.model` OllamaClient sends) must see the LOCAL model id when
286
+ // local backend is active, or a local QA session logs itself as the
287
+ // cloud model throughout even though it never touches OpenRouter.
288
+ const effectiveModel = useLocalBackend ? (resolvePerModeEnv("HEADLESSCODE_CODE_MODE_MODEL", mode) ?? model) : model
289
+ const client =
290
+ llmClient ??
291
+ (useLocalBackend
292
+ ? new OllamaClient({
293
+ baseUrl: baseUrl ?? resolvePerModeEnv("HEADLESSCODE_OLLAMA_URL", mode),
294
+ defaultModel: effectiveModel,
295
+ // OllamaClient has its own independent abort timer
296
+ // (ollama.ts's DEFAULT_OLLAMA_TIMEOUT_MS, 300s), separate
297
+ // from the llmTimeoutMs passed to HeadlessSession below —
298
+ // verified live 2026-08-28 (cli.ts's LOCAL_LLM_TIMEOUT_MS
299
+ // doc comment has the full story): raising the session-
300
+ // level timeout alone still left QA sessions dying at
301
+ // exactly 300000ms because this constructor never heard
302
+ // about it.
303
+ timeoutMs: useLocalBackend ? 630_000 : undefined,
304
+ })
305
+ : new OpenRouterClient({ apiKey: process.env.HEADLESSCODE_OPENROUTER_API_KEY, defaultModel: model, baseUrl }))
306
+
307
+ // Mirror every log line to <worktree>/qa.log — see reviewer.ts's runReview
308
+ // for the full rationale (same fix, same incident: review/QA run
309
+ // in-process rather than as a spawned subprocess with redirected stdout
310
+ // like a worker gets via harness.log, so there was no `tail -f`able file).
311
+ const logFilePath = path.join(workspaceRoot, "qa.log")
312
+ fs.appendFileSync(logFilePath, `\n===== headlesscode qa start: ${new Date().toISOString()} =====\n`, "utf-8")
313
+ const logger = new Logger({ level: "info", filePath: logFilePath })
314
+
315
+ const session = new HeadlessSession({
316
+ workspaceRoot,
317
+ mode,
318
+ model: effectiveModel,
319
+ taskText: options.taskText ?? defaultTaskText(workspaceRoot, mode),
320
+ maxIterations,
321
+ budget,
322
+ systemPromptOverride,
323
+ tools: qaTools(),
324
+ executor: createQaHeadlessExecutor(workspaceRoot),
325
+ llmClient: client,
326
+ // Issue #144 (mirrors reviewer.ts): local inference is free — a
327
+ // fabricated dollar figure in the QA log is noise at best.
328
+ trackCost: !useLocalBackend,
329
+ memory,
330
+ project,
331
+ logger,
332
+ // 2026-08-27 (mirrors reviewer.ts's runReview — same incident, same
333
+ // fix): without this, a local QA session's bare/fabricated text reply
334
+ // is accepted as an ordinary success, so a QA verdict parser has no
335
+ // way to tell "genuinely verified" from "session produced garbage" —
336
+ // exactly the gap that let a local reviewer session silently rubber-
337
+ // stamp a worker's false completion. Forcing explicit completion here
338
+ // makes a garbage QA reply a retried mistake / genuine session error
339
+ // instead of a silent false pass.
340
+ requireExplicitCompletion: useLocalBackend && !envBoolean("HEADLESSCODE_ALLOW_TEXT_ONLY_COMPLETION"),
341
+ // Same gap as cli.ts's LOCAL_LLM_TIMEOUT_MS (see its doc comment for
342
+ // the full story): this session construction never set llmTimeoutMs
343
+ // at all, so a QA session on the local daemon always used the
344
+ // generic DEFAULT_LLM_TIMEOUT_MS (300s) — shorter than the shim's
345
+ // own deliberately-raised 600s upstream patience, so the harness
346
+ // gives up first on a genuinely slow (not hung) local call.
347
+ llmTimeoutMs: useLocalBackend ? 630_000 : undefined,
348
+ // Same gap as cli.ts's LOCAL_MAX_TOKENS (see its doc comment for the
349
+ // full incident): unset here, a QA session on the local daemon falls
350
+ // back to loop.ts's DEFAULT_MAX_TOKENS (32768) — sized for a cloud
351
+ // reasoning model's 128K+ window, not this daemon's real
352
+ // 65,536-token total context, where one runaway generation can crash
353
+ // the whole session outright.
354
+ maxTokens: useLocalBackend ? 8192 : undefined,
355
+ })
356
+
357
+ const result: SessionResult = await session.run()
358
+ if (result.status !== "success" || result.result === undefined) {
359
+ const error = result.error ?? "unknown QA session error"
360
+ return {
361
+ verdict: "error",
362
+ evidence: "",
363
+ summary: `QA session failed: ${error}`,
364
+ }
365
+ }
366
+ return { ...parseQaResult(result.result), reportPath: result.reportPath }
367
+ }
368
+
369
+ /**
370
+ * Run QA, retrying ONLY when the session itself failed (verdict "error" —
371
+ * a crash, budget stop, bounded-failure mistake limit, or an
372
+ * attempt_completion whose result never actually populated), up to
373
+ * `maxRetries` additional attempts. A real "pass" or "fail" is returned
374
+ * immediately, first try.
375
+ *
376
+ * Mirrors reviewer.ts's runReviewWithRetries — same class of bug, same fix.
377
+ * Caught live 2026-08-05 running issue #18's own round: a QA session ended
378
+ * with no real result (`result.result === undefined` despite the
379
+ * HeadlessSession itself reporting "success"), correctly classified as
380
+ * verdict "error" by runQa — but `cli.ts` had nowhere to route that
381
+ * distinctly from a real "fail": the group's top-level status stayed "done",
382
+ * cost got recorded, and the round was silently considered fully settled
383
+ * with an empty-evidence "failed" QA that nobody would ever look at again.
384
+ */
385
+ export async function runQaWithRetries(options: RunQaOptions, maxRetries = 2): Promise<QaResult> {
386
+ let last: QaResult = { verdict: "error", evidence: "", summary: "no attempt made" }
387
+ for (let attempt = 0; attempt <= maxRetries; attempt++) {
388
+ last = await runQa(options)
389
+ if (last.verdict !== "error") {
390
+ return last
391
+ }
392
+ }
393
+ return last
394
+ }
395
+
396
+ /**
397
+ * Parse a QA session's final report into { verdict, evidence, summary }.
398
+ *
399
+ * Deterministic + tested:
400
+ * - "pass" — explicit pass markers (QA PASS, all tests pass, no errors
401
+ * found, definition of done satisfied, `errors: []`, …).
402
+ * - "fail" — explicit fail markers (QA FAIL, does not work, definition of
403
+ * done NOT met) — deliberately narrow, see below;
404
+ * - default "fail" when neither is stated (FAIL-CLOSED: an inconclusive QA
405
+ * report never gates a deploy through).
406
+ *
407
+ * The fail-marker list is deliberately narrow — the SAME class of bug as
408
+ * `reviewer.ts`'s parseReviewResult, found in the same incident: it used to
409
+ * include bare "failed"/"failure"/"failures found"/"errors found"/"bug"/
410
+ * "regression"/"broken"/"failing" ANYWHERE in the report. A QA report citing
411
+ * real baseline test counts ("12 failed, 1187 passed" — pre-existing on
412
+ * master, unrelated to the change) contains the bare word "failed", which
413
+ * alone tripped the fail branch — and fail always wins over pass in this
414
+ * function's logic, so ANY QA report mentioning ordinary baseline numbers
415
+ * would be force-classified as "fail" even when QA genuinely passed. The
416
+ * negative lookbehind `(?<!no\s)failed` only protects the exact phrase "no
417
+ * failed" — it does nothing for "12 failed", which is the actual shape
418
+ * baseline reporting takes. Narrowed to `qa fail` / `definition of done
419
+ * not` / `does not work` — explicit, structured verdict language, not
420
+ * incidental words that show up constantly in normal test-output prose.
421
+ * Do not add generic words back without a specific report shape that needs
422
+ * them AND a test proving no false-positive on baseline/prose language.
423
+ *
424
+ * Evidence: an `## Evidence` / `Evidence:` section when present, else the full
425
+ * summary text (which still contains the real command output).
426
+ */
427
+ export function parseQaResult(text: string): QaResult {
428
+ const summary = text.trim()
429
+ const lower = summary.toLowerCase()
430
+
431
+ // Extract an "## Evidence" / "Evidence:" section if present.
432
+ let evidence = ""
433
+ const section = summary.match(
434
+ /(?:^|\n)(?:#{1,6}\s*)?evidence\s*:?\s*\n([\s\S]*?)(?=\n#{1,6}\s|\n\s*(?:verdict|summary)\b|\s*$)/i,
435
+ )
436
+ if (section?.[1]) {
437
+ evidence = section[1].trim()
438
+ }
439
+
440
+ // Primary path: an explicit, structured "QA_VERDICT: PASS"/"QA_VERDICT: FAIL"
441
+ // line (required by defaultTaskText) is authoritative — exact match, no
442
+ // heuristics. Same rationale as reviewer.ts's parseReviewResult: free-form
443
+ // regex heuristics on prose have no ceiling on false positives (a real
444
+ // incident here — baseline "N failed" test counts forced a fail verdict on
445
+ // a QA session that explicitly said "QA PASS").
446
+ const structuredVerdict = summary.match(/^QA_VERDICT:\s*(PASS|FAIL)\s*$/im)
447
+ if (structuredVerdict) {
448
+ const verdict: QaVerdict = structuredVerdict[1]!.toUpperCase() === "PASS" ? "pass" : "fail"
449
+ return { verdict, evidence: evidence || summary, summary }
450
+ }
451
+
452
+ // Fallback for a session that didn't emit the required line.
453
+ //
454
+ // "QA PASS"/"QA FAIL" as a real verdict declaration must LEAD a line
455
+ // (like the structured QA_VERDICT: line above) -- not match anywhere in
456
+ // flowing prose. Verified live 2026-08-28: a QA report that correctly
457
+ // reported failure ("a QA PASS requires all checks to have run clean...
458
+ // I must fail rather than guess") got silently flipped to "pass"
459
+ // because the old bare `\bqa\s*pass\b` match fired on the substring
460
+ // "QA PASS" inside that EXPLANATORY sentence, which was never a real
461
+ // verdict declaration — and the actual conclusion ("I must fail")
462
+ // never matched `\bqa\s*fail\b` since it lacks the literal word "qa".
463
+ // Same bug class this file's own history already documents once (an
464
+ // over-eager FAIL match on "12 failed" baseline counts, see
465
+ // testBaselineFailedCountsDoNotFalselyFailAPassingQaReport) — this
466
+ // time in the opposite direction, an over-eager PASS match.
467
+ const qaPassLeadsLine = /^\s*(?:[#*-]\s*)*qa\s*pass\b/im.test(summary)
468
+ const qaFailLeadsLine = /^\s*(?:[#*-]\s*)*qa\s*fail\b/im.test(summary)
469
+ const hasPassMarker =
470
+ qaPassLeadsLine ||
471
+ /\b(all\s*tests?\s*pass|no\s+errors?\s+found|no\s+failures|definition\s+of\s+done\s+(is\s+|was\s+)?(met|satisfied)|errors?\s*:\s*\[\s*\]|passed\s+(\d+)\/\d+|verified\s+ok)\b/i.test(
472
+ lower,
473
+ )
474
+ const hasFailMarker = qaFailLeadsLine || /\b(definition\s+of\s+done\s+not|does\s+not\s+work)\b/i.test(lower)
475
+
476
+ let verdict: QaVerdict
477
+ if (hasPassMarker && !hasFailMarker) {
478
+ verdict = "pass"
479
+ } else if (hasFailMarker) {
480
+ verdict = "fail"
481
+ } else {
482
+ // Fail-closed: no explicit verdict → do not pass.
483
+ verdict = "fail"
484
+ }
485
+
486
+ return { verdict, evidence: evidence || summary, summary }
487
+ }