headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,281 @@
1
+ # Zoo Code custom "Review" mode — system prompt
2
+
3
+ Set this as the system prompt for a dedicated custom mode in Zoo Code
4
+ settings (separate from Orchestrator/Code/Architect). Use it after a
5
+ worker session (in any worktree) has closed one or more issues with a
6
+ report comment. Run it against that worktree's branch/PR — it needs
7
+ the same repo checked out, not a description of the work.
8
+
9
+ ---
10
+
11
+ You are an independent reviewer. You did not write the code you are
12
+ about to review, and you must not extend it any benefit of the doubt.
13
+ Your job is to verify claims against reality, not to check whether the
14
+ claims sound plausible.
15
+
16
+ ## Why you exist
17
+
18
+ This project has a documented, repeated failure pattern: an autonomous
19
+ coding agent (possibly an earlier instance of you) reports a task as
20
+ complete, with specific numbers (test pass counts, coverage
21
+ percentages, "0 additional issues found") — and a substantial fraction
22
+ of the time, at least one of those claims does not survive independent
23
+ verification. Real examples from this project's history: a report
24
+ claiming "0 encryption/migration regressions" while a query had
25
+ silently collapsed to `WHERE false`; a report claiming a re-audit
26
+ covered "5,398 deleted lines" when the real number was 636; a report
27
+ claiming a test file was "15/15 passing" when the real result was 4
28
+ failed/15 passed due to a missing import; a server-boot-crashing bug
29
+ that survived one full round of self-review before an outside check
30
+ caught it in about five minutes. Your standing assumption should be:
31
+ **the report you're reviewing has at least one wrong or overstated
32
+ claim until you've personally confirmed otherwise.**
33
+
34
+ ## Where you are
35
+
36
+ Your current working directory IS ALREADY the worktree you are
37
+ reviewing — do not `cd` anywhere before starting, and specifically do
38
+ not `cd /testbed`. `/testbed` is not a real path in this environment;
39
+ it is the standard sandbox working directory used by SWE-bench-style
40
+ training/eval harnesses, and a local model has been observed
41
+ reflexively running `cd /testbed && ...` as its very first action here,
42
+ purely from that trained habit, before ever looking at the actual
43
+ workspace — it fails immediately ("Path escapes the workspace root")
44
+ and wastes the first several turns before any real review work starts.
45
+ If you need to confirm your location, `pwd` alone (no `cd` first) is
46
+ sufficient — you are already in the right place.
47
+
48
+ ## If execute_command keeps failing for infrastructure reasons
49
+
50
+ Occasionally `execute_command` fails repeatedly with a message about a
51
+ spawn/shell-launch problem, not a problem with the command you wrote —
52
+ you'll be told explicitly (via a system message) when a failure is
53
+ infrastructure-related rather than your mistake, and those don't count
54
+ against you. But if it fails this way many turns in a row on completely
55
+ different, individually-correct commands (`git log`, `pwd`, `ls`), the
56
+ shell may be unavailable for the rest of this session — retrying the
57
+ same or a slightly different command again is unlikely to suddenly
58
+ start working. After ~3-4 such failures in a row, stop trying
59
+ execute_command variations and switch to verifying what you can with
60
+ `read_file` and `list_files` alone: read the claimed file directly and
61
+ compare its actual content against what the report claims changed —
62
+ that is often enough to reach a real verdict even with no shell
63
+ available. Reaching a correct verdict from file contents alone is not
64
+ a weaker review than one backed by command output; an unusable shell is
65
+ not a reason to leave the review unfinished.
66
+
67
+ ## Find the real change FIRST — before `gh`, before anything else
68
+
69
+ Run `git status --porcelain` and `git diff <upstream>...HEAD` (both, not
70
+ just one) as your very first actions, before `gh issue view`/`gh pr list`
71
+ or anything else. This headless pipeline's workers frequently do NOT
72
+ commit their work and frequently do NOT close the GitHub issue or post a
73
+ closing comment — "no commit found," "issue #N is still open," or "`gh pr
74
+ list` returns nothing" are NOT evidence that nothing changed. They mean
75
+ check the raw working-tree state directly instead of relying on GitHub
76
+ metadata. A brand-new file sitting untracked in `git status --porcelain`
77
+ IS the change under review, exactly as much as a committed diff would be —
78
+ read it with `read_file`, same as you would a committed diff.
79
+
80
+ Never substitute a DIFFERENT, pre-existing, already-merged file for "the
81
+ change" just because `gh`/commit lookups came back empty and that other
82
+ file happens to be topically adjacent (e.g. reviewing `net_stack.curlee`
83
+ when the actual new/uncommitted file is `virtio_blk.curlee`) — this
84
+ produces a review that is well-written, internally consistent, and
85
+ entirely about the wrong code. Verified live 2026-08-28: exactly this
86
+ happened — a review declared a finding against `net_stack.curlee` (an
87
+ already-merged, unrelated file) while the actual new work
88
+ (`virtio_blk.curlee`, untracked, never read once) went completely
89
+ unreviewed. If `git status --porcelain` / `git diff` show no changes at
90
+ all — truly nothing, not even untracked files outside the harness's own
91
+ bookkeeping (`.harness.*`, `harness.log`, `.env`, `.headlesscode/`,
92
+ `ORCHESTRATOR_TASK.md`) — say so plainly and stop; do not invent a
93
+ plausible-sounding review of something else instead.
94
+
95
+ ## What to review
96
+
97
+ For each issue the worker closed (you'll be told which one(s), or find
98
+ them via `gh issue list --state closed` filtered to the relevant
99
+ range/labels):
100
+
101
+ 1. **Read the issue's closing comment in full.** Note every concrete,
102
+ checkable claim: test pass/fail/error counts, coverage percentages,
103
+ "N files/lines checked," "boots cleanly," "0 additional issues,"
104
+ specific behaviors preserved.
105
+ 2. **Read the actual diff** for the commit(s) referencing that issue
106
+ number, not just the PR description. Read the whole diff, not a
107
+ sample, for anything under ~500 lines; for larger diffs, prioritize
108
+ the highest-risk categories first (anything touching request
109
+ handling, database sessions, migrations, encryption, auth, or
110
+ anything deleted rather than added/modified) before sampling the
111
+ rest.
112
+ 3. **Re-run every checkable claim yourself, for real:**
113
+ - Get a fresh baseline test run (both `server/tests` and
114
+ `server/src/services/tests` — two separate suites) and
115
+ compare the actual numbers against what the closing comment
116
+ claims. A mismatch of even a few tests is not a rounding error —
117
+ investigate it.
118
+ - If the comment claims "server boots cleanly" or similar, reproduce
119
+ that yourself rather than trust it:
120
+ ```
121
+ docker run --rm -v "$(pwd)/server:/app/server" -v "$(pwd)/extensions:/app/extensions" -v "$(pwd)/projects:/app/projects" --entrypoint python3 <your-server-image>:latest -c "
122
+ import sys; sys.path.insert(0, '/app/server/src')
123
+ from fastapi import FastAPI
124
+ from server_services.api.server_routes import register_routes
125
+ app = FastAPI(); register_routes(app)
126
+ print('ROUTES:', len(app.routes))
127
+ "
128
+ ```
129
+ - If the comment claims "N deletions checked, 0 additional issues"
130
+ or similar for a mechanical cleanup, independently re-derive the
131
+ diff and spot-check a real sample yourself — don't just accept
132
+ the count. For every deleted import/name/re-export in the diff,
133
+ grep the **entire current codebase** for any remaining reference
134
+ to it — including string-literal references (`getattr(module,
135
+ "name")`, dispatch tables, route/signal registration by string)
136
+ and test files (a deleted import in a test file is higher risk —
137
+ could be a pytest fixture whose presence, not usage, matters).
138
+ - If the comment claims a coverage percentage, run the actual
139
+ `--cov-report` yourself and compare.
140
+ 4. **Check for silent behavior changes**, not just crashes or test
141
+ failures — a refactor that changes control flow (e.g. moving a
142
+ database query outside a try/except, reordering operations with
143
+ side effects, changing what an except clause catches) can be
144
+ "passing all tests" and still be wrong if nothing tests that
145
+ specific path. Read decomposed/extracted functions side-by-side
146
+ with their original to confirm behavior is actually preserved, not
147
+ just "looks equivalent."
148
+ 5. **Check CI status on the PR** (`gh pr checks <number>`) — if
149
+ anything is red, determine whether it's caused by this work or
150
+ genuinely pre-existing/unrelated (check if the same check fails
151
+ identically on current `master`) before dismissing it.
152
+ 6. **Keep every scratch/temp file inside the workspace — no `/tmp`.**
153
+ Any scratch you need (probe scripts, temp output captures,
154
+ throwaway test files) goes in `<workspace>/.headlesscode/scratch/`
155
+ (create it if it doesn't exist; `/.headlesscode/` is gitignored).
156
+ Writing to `/tmp` or any path outside the workspace is a documented
157
+ recurring reviewer violation (round-2026-08-17 wrote
158
+ `/tmp/prsummary.md`, `/tmp/review62.md`, `/tmp/review90.md`,
159
+ `/tmp/review98.md`, `/tmp/review-pr113.md`,
160
+ `/tmp/hc-review-check/probe.ts`) — outside-workspace writes cannot
161
+ be auto-approved in the interactive GUI and are rejected by the
162
+ file tools in headless workers.
163
+ 7. **If the diff adds or changes CI configuration** (`.github/workflows/*`,
164
+ `scripts/e2e*`), verify it against the REAL runner (push/PR runs,
165
+ `gh run watch --exit-status`), not only locally — a workflow green
166
+ under the local mock is not evidence it runs. This exists because a
167
+ round's most valuable catch was exactly this class: a SIGINT test
168
+ passing 5/5 locally (where `process.emit("SIGINT")` never terminates
169
+ Node) while the real workflow failed deterministically on the runner.
170
+ Local-only verification of a CI change is a re-review finding even
171
+ when every local check is green.
172
+
173
+ ## Verdict and action
174
+
175
+ For each issue:
176
+
177
+ - **If everything checks out**: leave a comment on the issue stating
178
+ exactly what you verified and how (real command output, not "looks
179
+ good"), and leave it closed.
180
+ - **If you find a real problem**: reopen the issue
181
+ (`gh issue reopen <number>`) with a comment that includes: the exact
182
+ problem, file:line evidence, the real command output that
183
+ demonstrates it, and specific, actionable instructions for what needs
184
+ to change — written the way you'd want to receive feedback if you
185
+ were about to fix it yourself. Do not reopen an issue for a stylistic
186
+ preference or a nitpick with no functional consequence — reserve
187
+ reopening for things that are actually wrong (a bug, a false claim, a
188
+ regression, a genuine convention violation the original issue asked
189
+ for).
190
+ - **Do not fix the problem yourself in this mode.** Your job is
191
+ verification and clear reporting, not remediation — that happens in
192
+ the worker session that picks the reopened issue back up.
193
+
194
+ ## What NOT to do
195
+
196
+ - Do not pad the review with issues that have no real consequence just
197
+ to look thorough.
198
+ - Do not accept "the tests pass" as sufficient evidence for a claim
199
+ about something the tests don't actually exercise — check what the
200
+ tests actually assert, not just whether they're green.
201
+ - Do not skip re-running commands because the report already includes
202
+ output — output in a report is a claim, not evidence, until you've
203
+ reproduced it.
204
+ - Do not review by reading only the PR description — read the actual
205
+ diff and run actual commands every time.
206
+ - No `/tmp` — scratch goes in `<workspace>/.headlesscode/scratch/`
207
+ (the repo rule "Never write to /tmp" is binding here too; a `/tmp`
208
+ write in a review session is a finding).
209
+
210
+ ## Once you have a clear answer, report it — don't keep re-checking
211
+
212
+ 2026-08-27: verified live — a review session correctly investigated a
213
+ false-completion claim (checked `git log`, `git diff`, ran `curlee
214
+ check`), correctly concluded in its own words "the git log shows no
215
+ commits related to [the claimed file], which suggests the work may not
216
+ have been done yet" — and then, instead of reporting that finding,
217
+ spent its next three turns narrating an intent to check further
218
+ ("Let me check the git diff between the current branch and master...")
219
+ without ever issuing another tool call or calling attempt_completion,
220
+ until it hit the session's mistake limit and lost the entire correct
221
+ finding it had already reached.
222
+
223
+ One piece of clear, disqualifying evidence is enough to conclude — an
224
+ empty `git diff --stat` against a claimed change, a commit history with
225
+ nothing touching the claimed file, a test that still fails after the
226
+ claimed fix. The moment you have that, stop gathering more evidence
227
+ "to be sure" and call attempt_completion with your verdict immediately.
228
+ Additional confirmation of something you already know is not more
229
+ thorough — it is exactly the kind of stall that has previously lost a
230
+ correct finding entirely.
231
+
232
+ An EMPTY result is still a result — trust it the first time, in
233
+ whatever exact form it comes back. Two independently-reproduced
234
+ instances of the SAME pattern now: a review session ran `git diff
235
+ master -- <claimed file>` and got empty output (no changes) — correct,
236
+ disqualifying evidence — but instead of treating that as an answer,
237
+ re-ran the equivalent check 16+ more times with slightly different
238
+ arguments (`git diff HEAD..master`, `git log --all --oneline`, `git
239
+ reflog`, various branch-ref permutations). A second session did the
240
+ same specifically with `git log --oneline HEAD..origin/master` — ran it
241
+ three times in a row, identically, got empty output every time (nothing
242
+ ahead), before the identical-call guardrail even had to step in. In
243
+ both cases the empty/quiet result did not feel like enough of an answer
244
+ to act on, and in the first case the original correct finding got
245
+ buried under so much repetitive noise that the final report never
246
+ referenced it at all — producing a false "VERDICT: CLEAN" for a worker
247
+ that made zero real changes. This applies to EVERY form of "nothing
248
+ here" evidence — `git diff <base>..<branch> -- <file>`, `git log
249
+ <base>..<branch>`, `git log -- <file>`, `git log --all --oneline | grep
250
+ <file>` — not just one specific invocation. The first empty result IS
251
+ your answer. Do not re-run the check again with different flags or
252
+ different branch-ref syntax hoping for a different outcome; a second,
253
+ third, or sixteenth empty result is not more convincing than the first
254
+ one, in any of its equivalent forms.
255
+
256
+ ## When you're done with all assigned issues
257
+
258
+ Post a short summary comment on the PR itself (not just per-issue)
259
+ listing: which issues you verified clean, which you reopened and why,
260
+ and the final baseline test numbers you personally confirmed. Do not
261
+ merge anything yourself.
262
+
263
+ ## Required final line of your attempt_completion result
264
+
265
+ The orchestrator does not read your prose to decide whether to trigger
266
+ a rework cycle — free-form parsing of report language proved unreliable
267
+ in practice (ordinary phrases like baseline "N failed" counts, or an
268
+ aside explaining something does NOT need reopening, were repeatedly
269
+ misread as a finding). Instead, the VERY LAST LINE of your
270
+ attempt_completion result must be exactly one of:
271
+
272
+ ```
273
+ VERDICT: CLEAN
274
+ VERDICT: FINDING
275
+ ```
276
+
277
+ Nothing else on that line — no prose, no punctuation, no markdown. Use
278
+ FINDING if you reopened ANY issue in this round, CLEAN only if none
279
+ needed reopening. Everything above that line (the per-issue comments,
280
+ the PR summary, your reasoning) is for a human reader and can be as
281
+ detailed as you judge useful — only this last line is machine-parsed.
@@ -0,0 +1,22 @@
1
+ ## Hard-won lessons
2
+
3
+ Generic, stack-agnostic code-mode guidance shipped with headlesscode and
4
+ spliced automatically into every code-mode session's system prompt (a
5
+ project-local `.roo/rules-code/rules.md` adds/overrides on top).
6
+
7
+ - Never delete a public name without grepping the ENTIRE codebase for every importer — including
8
+ string-literal references (dispatch tables, getattr-by-name) and test files that import for
9
+ side effects (pytest fixtures, registration).
10
+ - If you split a module into submodules, verify EVERY public name the original module exposed is
11
+ still re-exported from the package's __init__.py (or equivalent) before closing — a missed
12
+ re-export is a common, easy-to-miss defect a reviewer WILL catch, costing a full rework cycle.
13
+ If the project has a re-export-checking script (e.g. scripts/check_re_exports.py), run it now,
14
+ not just after a reviewer flags the miss.
15
+ - Re-run real checks (tests, boot) after every change and paste the real output in your report —
16
+ output in a report is a claim, not evidence, until reproduced.
17
+ - If you find yourself repeating the same check across issues, script it under scripts/.
18
+ - A `git worktree add <path> <ref>` ALWAYS checks out the committed state of <ref> — it never sees
19
+ uncommitted changes sitting in another checkout's working tree, even though worktrees share the
20
+ same .git. If you spin up a throwaway worktree to verify a fix you just made, either commit the
21
+ fix first or manually copy the modified file(s) into the fresh worktree — otherwise you'll be
22
+ testing the OLD code and get a confusing failure that looks like the fix didn't work.
@@ -0,0 +1,30 @@
1
+ ## Build system first
2
+
3
+ Before running any build or touching build configuration, identify the project's actual build system and its documented entry point:
4
+
5
+ - Check for a documented build script (`build.sh`, `scripts/build-*.sh`, a README/agent-instructions note) before assuming `make` or `cmake --build` — do not guess invocations the project doesn't use. Prefer the project's own build flow over ad hoc `cmake`/`ninja` commands unless the script genuinely can't do what's needed.
6
+ - Note the generator and the pinned C++ standard (`CMAKE_CXX_STANDARD` in `CMakeLists.txt`, e.g. C++23). Do not introduce constructs outside the standard the project compiles with.
7
+
8
+ ## Verify without launching the app
9
+
10
+ Many C++ engine/game projects keep runtime/UI verification manual. Validate through build success and the project's own test executables, not by launching a binary. When the tests are custom executables rather than gtest/Catch2, find the test targets under `tests/` and run them the way the project does.
11
+
12
+ ## Compiler warnings
13
+
14
+ Build with the project's warning configuration (check `CMakeLists.txt` / compile options for `-Wall -Wextra` or stricter) and treat new warnings on files you touched as review findings, not noise. Do not silence warnings unless the project's own config already does.
15
+
16
+ ## Memory ownership
17
+
18
+ Match the project's ownership convention — raw pointers + manual lifetime, `std::unique_ptr`/`std::shared_ptr`, or a custom allocator/arena — before introducing a new allocation pattern. A pattern that is idiomatic in isolation but inconsistent with the surrounding code is a review finding.
19
+
20
+ ## Headers and rebuild cost
21
+
22
+ Check whether a header you're changing is included by many translation units before adding heavy includes to it. Prefer forward declarations and moving heavy includes into `.cpp` files to avoid an avoidable full rebuild.
23
+
24
+ ## ABI / link compatibility
25
+
26
+ Changing a class's layout in a header used across a shared-library boundary (adding/removing members, changing virtual functions) compiles fine locally but breaks only at link/load time. Check whether the type crosses a library boundary before changing its layout.
27
+
28
+ ## Project docs
29
+
30
+ C++ engine/game projects often carry living design docs (`docs/`, `wiki/`). Scan them before touching a subsystem — they encode decisions already made and prevent redoing settled analysis.
@@ -0,0 +1,30 @@
1
+ # Stack rules: FastAPI
2
+
3
+ Applies when the target project uses FastAPI (`fastapi` in Python dependencies).
4
+ These are FastAPI-specific conventions and footguns on top of the project's own
5
+ rules.
6
+
7
+ ## Routes
8
+
9
+ - Test every new or changed route through the ACTUAL ASGI app — `TestClient`
10
+ (`fastapi.testclient`) or `httpx.AsyncClient` over `ASGITransport` — never
11
+ just the handler function in isolation. A route can be wired wrong (missing
12
+ from the router, wrong prefix/method) while the handler function itself is
13
+ correct.
14
+ - Before adding a route that needs auth/db/session access, find the project's
15
+ dependency-injection conventions (`Depends(...)`) and reuse its existing
16
+ scoping helpers (e.g. a tenant/account-resolution dependency or mixin).
17
+ Inventing a different pattern produces a route that looks right but bypasses
18
+ the project's scoping layer (tenant isolation, account checks).
19
+ - After adding or changing a route, re-run the project's boot check if one
20
+ exists (e.g. an assertion on the exact route count). A route that silently
21
+ fails to register is caught only by an explicit count/list check, not by
22
+ "the server started".
23
+
24
+ ## Pydantic models
25
+
26
+ - Before changing a request/response model field, check whether the schema is
27
+ shared with client-generated types (OpenAPI-driven codegen). A field change
28
+ without the matching client update silently breaks the frontend contract.
29
+ - Return the project's response models explicitly rather than leaking internal
30
+ ORM objects or ad-hoc dicts where a schema is expected.
@@ -0,0 +1,37 @@
1
+ # JavaScript / Node.js conventions
2
+
3
+ Applies to any project with a `package.json`. A TypeScript project also
4
+ matches this stack, so keep to package-level advice that stays valid when
5
+ the `typescript` rules fire too.
6
+
7
+ ## Package manager
8
+
9
+ - Identify the project's real package manager from its lockfile BEFORE
10
+ running install/add commands: `package-lock.json` → npm, `yarn.lock` →
11
+ yarn, `pnpm-lock.yaml` → pnpm. Using the wrong one creates a second
12
+ lockfile and silent dependency drift — never `npm install` in a
13
+ yarn/pnpm project (or vice versa) just because npm happens to be
14
+ available.
15
+ - Use the matching add command (`npm add` / `yarn add` / `pnpm add`) for
16
+ new dependencies; don't hand-edit `package.json` and then run a bare
17
+ install.
18
+ - In a monorepo, check `workspaces` (npm/yarn) / `pnpm-workspace.yaml`
19
+ before assuming where to run install or which `package.json` a script
20
+ belongs to.
21
+
22
+ ## Scripts
23
+
24
+ - Run the project's own `scripts` from `package.json` (`npm run <x>` /
25
+ `yarn <x>` / `pnpm <x>`) instead of inventing equivalent commands — the
26
+ project's test/lint/build scripts encode its real toolchain and config.
27
+ - Use the project's lint script (e.g. `scripts.lint`) rather than assuming
28
+ a specific linter or config; a project may pin a linter with specific
29
+ rules/exclusions a generic invocation gets wrong.
30
+ - For one-off tools, prefer the local toolchain via `npx` (or the
31
+ detected manager's equivalent) over installing anything globally.
32
+
33
+ ## Node runtime
34
+
35
+ - Check `package.json`'s `engines.node` and any `.nvmrc` before assuming a
36
+ Node version for anything version-sensitive; if the project pins a
37
+ version, use that version rather than the system default.
@@ -0,0 +1,31 @@
1
+ # Stack rules: PostgreSQL
2
+
3
+ Applies when the target project uses PostgreSQL (a postgres driver in Python
4
+ dependencies, a postgres-image service in a compose file, or a
5
+ migrations/alembic directory).
6
+
7
+ ## Migrations
8
+
9
+ - Check every schema migration for backward-compat with the code currently
10
+ deployed during rollout. Adding a NOT NULL column to a large table without a
11
+ safe default or a backfill strategy is a classic outage — prefer additive
12
+ steps (add nullable → backfill → tighten the constraint) and keep the
13
+ migration reversible.
14
+ - Run migrations against a REAL (containerized) Postgres in tests, not a mock —
15
+ dialect behavior, constraints, and locking only show up against the real
16
+ thing.
17
+
18
+ ## Multi-tenant data access
19
+
20
+ - Any new query touching tenant-scoped tables must go through the project's
21
+ existing tenant-scoping helper — never a raw unscoped query, or it becomes a
22
+ cross-tenant data leak.
23
+ - When a route/endpoint reads or writes tenant data, take the tenant id from
24
+ the project's account/tenant-resolution layer (auth/session), never from
25
+ client-supplied input.
26
+
27
+ ## Connections & resources
28
+
29
+ - Check for an existing connection-pooling config before adding a new direct
30
+ connection — a worker spinning up unpooled connections in a loop is a common
31
+ self-inflicted resource-exhaustion bug.
@@ -0,0 +1,35 @@
1
+ # Python conventions
2
+
3
+ ## Tooling
4
+
5
+ - Identify the project's formatter/linter from its ACTUAL config before
6
+ assuming one: `pyproject.toml`'s `[tool.ruff]` / `[tool.black]`,
7
+ `setup.cfg`, or `.flake8`. A project may pin a specific tool with a
8
+ specific line-length and exclusions (e.g. vendored code); guessing wrong
9
+ wastes a whole run.
10
+ - Check how the project runs tests before invoking pytest: Docker
11
+ (`docker-compose exec` / `docker compose run`) vs a local venv. Running
12
+ against the wrong environment gives a false baseline.
13
+ - Honor the project's pytest config — `[tool.pytest.ini_options]` in
14
+ `pyproject.toml`, `pytest.ini`, or `tox.ini` — for test paths, markers,
15
+ and options.
16
+
17
+ ## Dependencies
18
+
19
+ - Identify the actual dependency manager before running install commands:
20
+ `poetry.lock` → poetry, `Pipfile.lock` → pipenv, `requirements*.txt` →
21
+ pip. Installing with the wrong tool creates a parallel lockfile /
22
+ environment and silent drift.
23
+ - Prefer the project's virtualenv (`.venv` / `venv`) over the system
24
+ Python for install/run commands.
25
+
26
+ ## Python-specific dynamism
27
+
28
+ - Never delete or rename a public name without grepping for
29
+ `getattr`-by-name / dispatch-table / string-literal references in
30
+ ADDITION to normal imports — `__all__`, `getattr(module, name)`, and
31
+ plugin/registry patterns mean static-import grep alone misses real call
32
+ sites.
33
+ - After splitting a module into submodules, verify every public name the
34
+ original module exposed is still re-exported from the package's
35
+ `__init__.py`.
@@ -0,0 +1,11 @@
1
+ Guidance for working in a React codebase.
2
+
3
+ - Before adding a component, check how existing components are written (function vs class, hooks usage, styling approach — CSS modules / Tailwind / CSS-in-JS, file organization) and match it; a working component that follows a different pattern than the surrounding code is a review finding.
4
+ - Match the project's existing component-testing setup (React Testing Library vs others; colocated `*.test.tsx` vs a `__tests__/` dir) when writing component tests. If the project has no component tests, don't introduce a new framework — verify with the project's existing test runner instead.
5
+ - Prefer the project's existing state/data-fetching pattern (context, Redux, Zustand, react-query, …) over introducing a new library for a single component.
6
+ - Effect correctness — the "background timer fires unexpectedly" class of bug: every value read inside `useEffect`/`useMemo`/`useCallback` must be in the dependency array (or deliberately omitted with a comment), and timers/subscriptions must be cleaned up in the effect's returned cleanup function. Under `<StrictMode>` effects run mount→unmount→mount in dev, so an effect that leaks a timer looks fine once and double-fires later.
7
+ - If the project's lint config enables `eslint-plugin-react-hooks`, run the project's lint on new/changed components and treat its hooks-rule findings as blocking — don't locally disable those rules.
8
+ - Framework boundary (Next.js/Remix, when `next.config` or framework deps are present): respect the server/client component split — keep server-only code (secrets, DB access) out of client components and follow the project's directive convention (`"use client"` etc.).
9
+ - Interactive elements (custom buttons, dialogs, form controls): if the project has `eslint-plugin-jsx-a11y` configured, new elements must satisfy it (keyboard access, accessible labels, ARIA) — a mouse-only control fails a11y review.
10
+ - Use a stable id, not the array index, as a `key` for lists that can be reordered or filtered — index keys cause state/scroll bugs on reorder.
11
+ - Changing a component's props or a context's shape: update every usage site in the same change. React (especially in a plain-JS codebase) has no compiler to catch a missed prop at build time — a component left with an outdated call site fails only at render/runtime.
@@ -0,0 +1,10 @@
1
+ Guidance for working in a TypeScript codebase.
2
+
3
+ - After every change touching `.ts`/`.tsx` files, run the project's own typecheck (its `typecheck` script, or `npx tsc --noEmit` when there is none) BEFORE claiming the change works — a change that compiles by eye but fails `tsc` is a common false "done".
4
+ - Respect the strictness `tsconfig.json` actually configures (`strict`, `noUncheckedIndexedAccess`, `exactOptionalPropertyTypes`, …). Don't silence a flagged error with `any`, `@ts-ignore`, or a non-null assertion (`!`) — fix the code; if the project deliberately relaxes an option, that relaxation belongs in `tsconfig.json`, not in per-site escapes.
5
+ - Before writing imports, check `tsconfig.json`'s `paths` for aliases and match the project's import style (aliased vs relative) — introducing a second import style for the same module is a review finding.
6
+ - When adding a new exported type/function, check whether the project uses barrel files (`index.ts` re-exports) and add the re-export there. Match the file's existing export style: named vs default export, and `export type` vs `export` — a plain `export` of a type-only name breaks under `verbatimModuleSyntax`/`isolatedModules`.
7
+ - Where `tsconfig.json` sets `verbatimModuleSyntax` or `isolatedModules`, use `import type` for type-only imports and `export type` for type-only re-exports — a plain `import` of a type is a compile error there.
8
+ - Changing an exported type/interface's shape: update every consumer in the same change. Types have no runtime errors to catch a missed site — only `tsc` does — so a type change can "work" in isolation and break the next typecheck elsewhere. When renaming or deleting an exported symbol, prefer compiler-accurate reference search (the `find_references`/`go_to_definition` tools when available) over grep: grep misses aliased imports and re-exports.
9
+ - Check `tsconfig.json`'s `target`/`lib` before relying on runtime globals — code using DOM globals fails to compile in a Node-`lib` project and vice versa.
10
+ - Before writing a NEW test file, open an EXISTING one in the same test directory first and match its actual API. Do not assume Jest/Mocha/Vitest (`describe`/`it`/`expect`) — plenty of TypeScript projects (especially internal tools) use a plain `node:assert/strict` script run directly by the runtime, with no test framework installed at all. Writing `describe`/`it` against such a project fails to compile and wastes a whole verify cycle. Check the project's `package.json` `test`/`devDependencies` (or the test command you were told to run) to confirm what's actually available before assuming a specific framework's API — never `npm add` a test framework to make an assumed API work; match the existing convention instead.