headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,2575 @@
1
+ /**
2
+ * Headless tool executor.
3
+ *
4
+ * Implements the Phase 1 core tools using plain `fs` + `child_process` — NO
5
+ * `vscode.*` anywhere. Argument names match the vendored native-tool schemas
6
+ * exactly (see `src/vendor/zoo-code/src/core/prompts/tools/native-tools/`):
7
+ *
8
+ * - read_file { path, offset?, limit? }
9
+ * - write_to_file { path, content }
10
+ * - apply_diff { path, diff }
11
+ * - search_replace { file_path, old_string, new_string }
12
+ * - edit_file { file_path, old_string, new_string, expected_replacements? }
13
+ * - execute_command { command, cwd?, timeout? }
14
+ * - list_files { path?, recursive? }
15
+ * - browser_action { action, url?, selector?, text? } — Playwright-backed
16
+ * headless browser inspection (launch/screenshot/click/type/getConsoleLogs/
17
+ * getNetworkErrors/close); NEW work, no upstream port (see src/tools/browser/).
18
+ * - outline / go_to_definition / find_references / import_graph — the four
19
+ * code-intelligence tools (TS Compiler API backed, src/codeintel/): one
20
+ * cached ts.Program per workspace serves all four. NEW work, no upstream
21
+ * port (see src/codeintel/).
22
+ *
23
+ * Every file operation resolves relative to the configured workspace root and
24
+ * is rejected if it escapes the workspace — both lexically (`../` traversal)
25
+ * and after following symlinks (a symlink whose real location is outside the
26
+ * workspace is refused; see resolveWithinWorkspace). Results are plain strings
27
+ * + an { isError } flag; long outputs are truncated to keep the model context
28
+ * bounded.
29
+ *
30
+ * The three surgical edit tools are backed by the vendored Zoo Code diff
31
+ * logic: `apply_diff` uses the fuzzy MultiSearchReplaceDiffStrategy
32
+ * (src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts),
33
+ * `search_replace` is a strict one-occurrence literal match, and `edit_file`
34
+ * falls back exact → whitespace-tolerant → token-based matching (plus
35
+ * file-creation when old_string is "").
36
+ *
37
+ * read_file carries a session-scoped cache (see the readFileCache comment
38
+ * below): the exact same effective args served against byte-identical content
39
+ * earlier in THIS session get a short cache-hit message instead of the full
40
+ * content, because re-sending identical content is pure output-token waste.
41
+ *
42
+ * OPT-IN local summarization (see src/tools/output-summarizer.ts): when
43
+ * HEADLESSCODE_LOCAL_SUMMARIZATION=1, oversized execute_command output that
44
+ * would exceed MAX_RESULT_CHARS is compressed by a small local Ollama model
45
+ * before reaching the cloud model (with a "[Output summarized by local
46
+ * model…]" transparency header). OFF by default; on ANY failure the result
47
+ * falls back to today's exact blunt truncation. Deliberately limited to
48
+ * execute_command output — read_file / write_to_file / diff content always
49
+ * stays verbatim (summarizing code a model is about to edit would be a
50
+ * correctness hazard).
51
+ *
52
+ * KNOWN LIMITATION (issue #88) — edit-tool read-modify-write is NOT atomic:
53
+ * `apply_diff`, `search_replace`, `edit_file`, and `write_to_file` each read
54
+ * the current file, compute the new content, and write it back with a plain
55
+ * `fs.writeFile` — no mtime re-check, no lock, no compare-and-swap. Within
56
+ * one session this is safe: the loop serializes edits to the same file and
57
+ * restricts the parallel read-only tool-call group to paths that aren't
58
+ * being edited (see loop.ts's same-file batching / parallel-group
59
+ * restriction). It is NOT safe across an external editor or a second,
60
+ * concurrent headlesscode session touching the same file — whichever write
61
+ * lands last silently wins and the other one's changes are lost (classic
62
+ * TOCTOU). No fix is planned; this is accepted as a known limitation rather
63
+ * than adding locking complexity for a single-session tool.
64
+ */
65
+
66
+ import * as fs from "node:fs"
67
+ import * as fsp from "node:fs/promises"
68
+ import * as path from "node:path"
69
+ import { createHash } from "node:crypto"
70
+ import { spawn, type ChildProcess } from "node:child_process"
71
+ import { setTimeout as sleep } from "node:timers/promises"
72
+
73
+ import { isTypeScriptWorkspace } from "./language-detect.js"
74
+ import type { Logger } from "../engine/logger.js"
75
+
76
+ // Side effect: installs String.prototype.toPosix() used by path formatting.
77
+ import "../vendor/zoo-code/src/utils/path.js"
78
+
79
+ import { MultiSearchReplaceDiffStrategy } from "../vendor/zoo-code/src/core/diff/strategies/multi-search-replace.js"
80
+
81
+ import { browserActionHandler, disposeBrowserSessions } from "./browser/handler.js"
82
+ import { describeImageHandler } from "../vision/tool.js"
83
+ import {
84
+ goToDefinitionHandler,
85
+ findReferencesHandler,
86
+ importGraphHandler,
87
+ outlineHandler,
88
+ renameSymbolHandler,
89
+ } from "../codeintel/handlers.js"
90
+ import { runTestsHandler } from "./run-tests.js"
91
+
92
+ import type { AuxLlmUsage, ToolContext, ToolHandler, ToolResult } from "../engine/types.js"
93
+ import {
94
+ checkCommand,
95
+ checkRedirectEscape,
96
+ describeRedirect,
97
+ type CommandRefusal,
98
+ type RedirectTarget,
99
+ } from "../permissions/commands.js"
100
+ import {
101
+ OllamaOutputSummarizer,
102
+ isLocalSummarizationEnabled,
103
+ summarizeToolResult,
104
+ MAX_SUMMARIZER_INPUT_CHARS as SUMMARIZER_INPUT_CAP,
105
+ } from "./output-summarizer.js"
106
+ import { findMatchingPattern } from "../permissions/protected-files.js"
107
+ import { resolvePermissions, type PermissionsConfig } from "../permissions/config.js"
108
+ import { createEmbedder, EMBEDDING_BACKEND_ENV, resolveEmbeddingBackend } from "../codesearch/embedder.js"
109
+ import { indexFilePath, loadIndexMetadata } from "../codesearch/index.js"
110
+ import { formatSearchResults, searchIndex } from "../codesearch/search.js"
111
+
112
+ /**
113
+ * Absolute path to bash, if present, for execute_command's shell (see the
114
+ * spawn() call below) — `undefined` falls back to `spawn`'s own default
115
+ * (`/bin/sh`) on a host without bash, rather than failing to spawn at all.
116
+ *
117
+ * That fallback must never be SILENT: it reintroduces the exact class of bug
118
+ * fixed in issue #35 (bash-only syntax like `${PIPESTATUS[0]}` silently
119
+ * failing the whole command line under `/bin/sh`, which produced a real
120
+ * false QA_VERDICT: FAIL). A future environment (e.g. a minimal Alpine-based
121
+ * Docker image shipping only `ash`) that lacks bash would quietly bring this
122
+ * back with no signal — so warn loudly, once, at module load, instead of
123
+ * letting it fail silently again.
124
+ */
125
+ export const BASH_PATH: string | undefined = fs.existsSync("/bin/bash")
126
+ ? "/bin/bash"
127
+ : fs.existsSync("/usr/bin/bash")
128
+ ? "/usr/bin/bash"
129
+ : undefined
130
+
131
+ if (BASH_PATH === undefined) {
132
+ process.stderr.write(
133
+ "[executor] WARNING: bash not found (checked /bin/bash, /usr/bin/bash) — execute_command falls back to " +
134
+ "/bin/sh, which does NOT support bash-only syntax (${PIPESTATUS[0]}, [[ ]], arrays). This previously " +
135
+ "caused a real false QA_VERDICT: FAIL (issue #35). Install bash in this environment to avoid it.\n",
136
+ )
137
+ }
138
+
139
+ /** Cap on tool-result text fed back to the model (keep context bounded). */
140
+ export const MAX_RESULT_CHARS = 30_000
141
+
142
+ /**
143
+ * Per-stream accumulation cap for execute_command output. The small margin
144
+ * above MAX_RESULT_CHARS guarantees the final combined string always exceeds
145
+ * the cap, so truncate()'s "output truncated" trailer still fires (a combined
146
+ * string exactly at the cap would be returned unchanged, silently losing it).
147
+ */
148
+ const MAX_COMMAND_STREAM_CHARS = MAX_RESULT_CHARS + 1024
149
+
150
+ /**
151
+ * Default read_file slice-mode line limit when the model passes no explicit
152
+ * `limit`. 600 lines, down from the vendored tool-schema default of 2000: a
153
+ * no-arg broad read historically pulled up to ~30k chars (~7-8k tokens) of
154
+ * history per call, and the truncation header already tells the model to page
155
+ * with `offset` for anything bigger. An explicit `limit` arg always wins, so
156
+ * 2000 stays available as an opt-in. Measured from real session logs: ~29% of
157
+ * read_file calls were no-arg broad reads.
158
+ */
159
+ export const DEFAULT_READ_LIMIT = 600
160
+
161
+ /** Resolve the read_file slice-mode default limit (env-overridable). */
162
+ export function readLimitFromEnv(env: NodeJS.ProcessEnv = process.env): number {
163
+ const raw = env.HEADLESSCODE_READ_LIMIT
164
+ if (raw === undefined || raw === "") {
165
+ return DEFAULT_READ_LIMIT
166
+ }
167
+ const n = Number(raw)
168
+ return Number.isInteger(n) && n > 0 ? n : DEFAULT_READ_LIMIT
169
+ }
170
+
171
+ /** Default execute_command timeout in seconds (matches vendored default ~120s). */
172
+ export const DEFAULT_COMMAND_TIMEOUT_S = 120
173
+
174
+ /** Max entries returned by list_files before truncation. */
175
+ export const MAX_LIST_FILES = 500
176
+
177
+ /** Options that configure a ToolExecutor's decision-escalation behavior. */
178
+ export interface ToolExecutorOptions {
179
+ /** ask_followup_question escalation timeout, ms (default 30 min — see DEFAULT_DECISION_TIMEOUT_MS). */
180
+ decisionTimeoutMs?: number
181
+ /** ask_followup_question poll interval, ms (default 5s — override in tests). */
182
+ decisionPollIntervalMs?: number
183
+ /**
184
+ * Resolved permissions (command allow/deny + protected files). When absent
185
+ * the executor resolves them itself from env vars +
186
+ * `<workspaceRoot>/.headlesscode/permissions.json` + built-in defaults, so
187
+ * executors constructed without CLI flags (reviewer/QA, tests) still
188
+ * enforce the repo's committed policy.
189
+ */
190
+ permissions?: PermissionsConfig
191
+ /** See ToolContext.guardLargeOverwrites (types.ts) for the full writeup. */
192
+ guardLargeOverwrites?: boolean
193
+ /**
194
+ * Live worker monitoring: fired at the same lifecycle points where the
195
+ * `.harness.needs-decision` marker is written/cleared, so the session can
196
+ * mirror them on its event feed (see ToolContext.onDecisionEvent).
197
+ */
198
+ onDecisionEvent?: (eventType: "decision_blocked" | "decision_answered", fields: Record<string, unknown>) => void
199
+ /**
200
+ * Live todo-list monitoring: fired each time update_todo_list replaces the
201
+ * session's checklist (see ToolContext.onTodoEvent). The session mirrors
202
+ * it on its event feed as `todo_updated`.
203
+ */
204
+ onTodoEvent?: (fields: { todos: string; done: number; inProgress: number; pending: number }) => void
205
+ /**
206
+ * Test-selection (run_tests): optional session-provided inference of the
207
+ * files changed since the session's baseline. Wired by HeadlessSession to
208
+ * diff the shadow-checkpoint repo's baseline commit against the current
209
+ * working tree (see src/checkpoints/service.ts) — the checkpoint service
210
+ * tracks a baseline per task, so this works even when the workspace
211
+ * itself has no git repo. Returning `undefined` makes the run_tests
212
+ * handler fall back to the workspace's own `git status`. Absent for bare
213
+ * executors (tests, reviewer/QA — which don't register run_tests anyway).
214
+ */
215
+ getSessionChangedFiles?: () => Promise<string[] | undefined>
216
+ /**
217
+ * Auxiliary LLM usage reporting (cloud vision captioning — see
218
+ * src/vision/describe.ts). Wired by HeadlessSession right after it
219
+ * constructs a BudgetTracker (same place as setBudgetClockHooks), so every
220
+ * captioning call's tokens/cost lands in the SAME BudgetTracker + running
221
+ * session totals as a main call — never an untracked side channel. Absent
222
+ * for bare executors (tests, reviewer/QA) — which also disables the
223
+ * screenshot action's automatic captioning, so an un-accounted executor
224
+ * never spends money behind the session's back (the model can still call
225
+ * `describe_image` explicitly).
226
+ */
227
+ onAuxLlmUsage?: (usage: AuxLlmUsage) => void
228
+ }
229
+
230
+ export class ToolExecutor {
231
+ private readonly handlers = new Map<string, ToolHandler>()
232
+ private pauseBudgetClock: (() => void) | undefined
233
+ private resumeBudgetClock: (() => void) | undefined
234
+ private onAuxLlmUsage: ((usage: AuxLlmUsage) => void) | undefined
235
+ /**
236
+ * Session-scoped read_file cache (see readFileCache below). One executor
237
+ * serves one session, so per-instance state is exactly per-session state.
238
+ * Keyed by a serialized string (see makeReadFileCacheKey) — the map must
239
+ * never be keyed by object identity, or no two calls would ever collide.
240
+ */
241
+ private readonly readCache = new Map<string, ReadFileCacheEntry>()
242
+ /**
243
+ * Session-scoped todo list state (see TodoListState below). update_todo_list
244
+ * always REPLACES the whole checklist, so this holds only the latest one.
245
+ * Conversational/session state — NEVER written to a workspace file.
246
+ */
247
+ private readonly todoList = new TodoListState()
248
+ /**
249
+ * Session-scoped repeat-call guard for list_files (see listFilesHandler's
250
+ * doc comment). Keyed by the call's effective (path, recursive) — value is
251
+ * the condensationGeneration this key was last listed at, so a repeat is
252
+ * only refused when nothing has been condensed since (the earlier result
253
+ * might have been evicted from context, in which case re-listing is the
254
+ * only way to see it again).
255
+ */
256
+ private readonly listFilesCalls = new Map<string, ListFilesCallEntry>()
257
+ /**
258
+ * Incremented by notifyCondensed() every time HeadlessSession applies a
259
+ * condensation (sync or background) — see listFilesCalls above.
260
+ */
261
+ private condensationGeneration = 0
262
+ /** Resolved permissions handed to every handler call (see ToolExecutorOptions.permissions). */
263
+ readonly permissions: PermissionsConfig
264
+
265
+ constructor(
266
+ readonly workspaceRoot: string,
267
+ private readonly options: ToolExecutorOptions = {},
268
+ ) {
269
+ this.permissions =
270
+ options.permissions ?? resolvePermissions({ workspaceRoot: this.workspaceRoot, env: process.env })
271
+ this.onAuxLlmUsage = options.onAuxLlmUsage
272
+ }
273
+
274
+ register(name: string, handler: ToolHandler): void {
275
+ this.handlers.set(name, handler)
276
+ }
277
+
278
+ has(name: string): boolean {
279
+ return this.handlers.has(name)
280
+ }
281
+
282
+ names(): string[] {
283
+ return [...this.handlers.keys()]
284
+ }
285
+
286
+ /**
287
+ * Wired by HeadlessSession right after it constructs a BudgetTracker, so
288
+ * ask_followup_question (and any future blocking tool) can pause the
289
+ * session's budget-duration clock while waiting on an external answer.
290
+ * Never called when no budget is configured.
291
+ */
292
+ setBudgetClockHooks(pause: () => void, resume: () => void): void {
293
+ this.pauseBudgetClock = pause
294
+ this.resumeBudgetClock = resume
295
+ }
296
+
297
+ /**
298
+ * Register read_file wired to THIS executor's session-scoped cache. Must be
299
+ * called by every construction path that wants the cache (headless,
300
+ * reviewer, QA) — each executor instance gets its own independent cache.
301
+ */
302
+ registerReadFile(): void {
303
+ this.register("read_file", (args, ctx) =>
304
+ readFileHandler(args, ctx, this.readCache.get(readFileKey(args, ctx)), this.readCache),
305
+ )
306
+ }
307
+
308
+ /**
309
+ * Register list_files wired to THIS executor's session-scoped repeat-call
310
+ * guard (see listFilesCalls above and listFilesHandler's doc comment).
311
+ * Every construction path that offers list_files should call this instead
312
+ * of a bare `register("list_files", listFilesHandler)`.
313
+ */
314
+ registerListFiles(): void {
315
+ this.register("list_files", (args, ctx) => listFilesHandler(args, ctx, this.listFilesCalls, this.condensationGeneration))
316
+ }
317
+
318
+ /**
319
+ * Called by HeadlessSession right after it splices a condensation into the
320
+ * live history (both the synchronous and background paths) — advances the
321
+ * generation the list_files repeat-call guard checks against, so a call
322
+ * repeated after a condensation is allowed again instead of refused.
323
+ */
324
+ notifyCondensed(): void {
325
+ this.condensationGeneration++
326
+ }
327
+
328
+ /**
329
+ * Register update_todo_list wired to THIS executor's session-scoped todo
330
+ * state, surfacing each state change via the onTodoEvent option (the
331
+ * session mirrors it as a `todo_updated` feed event). Must be called by
332
+ * the headless construction path; read-only executors (reviewer/QA/local
333
+ * explore) deliberately leave it unregistered — their tool lists never
334
+ * advertise it either, so it stays an inert stub there.
335
+ */
336
+ registerTodoList(): void {
337
+ this.register("update_todo_list", (args, ctx) => updateTodoListHandler(args, ctx, this.todoList))
338
+ }
339
+
340
+ /**
341
+ * Snapshot of the session's current todo list, or undefined before the
342
+ * first update_todo_list call. Read-only accessor — the dashboard /
343
+ * observability side can query live planning state without touching it.
344
+ */
345
+ getTodoList(): TodoListSnapshot | undefined {
346
+ return this.todoList.snapshot()
347
+ }
348
+
349
+ async execute(name: string, args: Record<string, unknown>): Promise<ToolResult> {
350
+ const handler = this.handlers.get(name)
351
+ if (!handler) {
352
+ return {
353
+ content: `[Error] Unknown tool: ${name}. This harness has no handler registered for it.`,
354
+ isError: true,
355
+ }
356
+ }
357
+ try {
358
+ return await handler(args, {
359
+ workspaceRoot: this.workspaceRoot,
360
+ permissions: this.permissions,
361
+ guardLargeOverwrites: this.options.guardLargeOverwrites,
362
+ decisionTimeoutMs: this.options.decisionTimeoutMs,
363
+ decisionPollIntervalMs: this.options.decisionPollIntervalMs,
364
+ pauseBudgetClock: this.pauseBudgetClock,
365
+ resumeBudgetClock: this.resumeBudgetClock,
366
+ onDecisionEvent: this.options.onDecisionEvent,
367
+ onTodoEvent: this.options.onTodoEvent,
368
+ // Auxiliary LLM usage (cloud vision captioning): forwarded so
369
+ // handlers can report spend into the session's BudgetTracker.
370
+ onAuxLlmUsage: this.onAuxLlmUsage,
371
+ })
372
+ } catch (err) {
373
+ return {
374
+ content: `[Error] Tool '${name}' failed: ${err instanceof Error ? err.message : String(err)}`,
375
+ isError: true,
376
+ }
377
+ }
378
+ }
379
+
380
+ /**
381
+ * Session teardown: hard-kill any execute_command children that a timeout
382
+ * left running in the background (see executeCommandHandler), AND close
383
+ * any launched browser (see disposeBrowserSessions — a model may never
384
+ * call browser_action's close(), so the browser is torn down here at true
385
+ * session end, exactly like the backgrounded children). A single tool
386
+ * call timing out mid-session does NOT trigger this — the backgrounded
387
+ * process is exactly what the model asked for and must keep running until
388
+ * the model cleans it up or the session truly ends. Only calling this at
389
+ * true session end does, so nothing is orphaned past its session.
390
+ */
391
+ dispose(): void {
392
+ killBackgroundCommands()
393
+ disposeBrowserSessions()
394
+ }
395
+ }
396
+
397
+ // ─── Path safety ─────────────────────────────────────────────────────────────
398
+
399
+ export class PathTraversalError extends Error {
400
+ constructor(requested: string, root: string) {
401
+ super(`Path escapes the workspace root (${root}): ${requested}`)
402
+ this.name = "PathTraversalError"
403
+ }
404
+ }
405
+
406
+ /**
407
+ * A path whose REAL location (after following symlinks) escapes the workspace
408
+ * even though its lexical path stays inside it (issue #64). Subclasses
409
+ * PathTraversalError so existing callers that catch the base class (e.g. the
410
+ * dashboard's HTTP layer, which maps it to a 400) keep treating it the same
411
+ * way.
412
+ */
413
+ export class SymlinkEscapeError extends PathTraversalError {
414
+ constructor(requested: string, root: string, resolvedTo: string) {
415
+ super(requested, root)
416
+ this.name = "SymlinkEscapeError"
417
+ this.message = `Path escapes the workspace root through a symlink (${root}): ${requested} — it resolves to ${resolvedTo}`
418
+ }
419
+ }
420
+
421
+ /**
422
+ * Resolve `p` against the workspace root and reject anything that escapes it.
423
+ *
424
+ * Two layers of containment (issue #64):
425
+ * 1. Lexical: path.resolve + a prefix check — the classic `../` traversal
426
+ * guard. This alone is NOT sufficient: every `fsp` call in the file tools
427
+ * FOLLOWS symlinks, so an agent can `ln -s ~/.ssh <ws>/sshlink` and then
428
+ * read AND write outside the workspace through it. The protected-files
429
+ * guard is bypassed the same way — it only ever sees the
430
+ * workspace-relative lexical path.
431
+ * 2. Symlink-following (assertNoSymlinkEscape): the real path of the deepest
432
+ * resolvable ancestor must stay inside the workspace root's own real
433
+ * path. An escaping symlink — as a directory component, as the target
434
+ * itself, or as a dangling symlink a later write would create THROUGH —
435
+ * is refused with SymlinkEscapeError. The ancestor walk keeps writes to
436
+ * not-yet-existing paths working (realpath on a nonexistent path throws).
437
+ *
438
+ * Documented boundary: a path is usable only if every existing component of
439
+ * it really lives inside the workspace. A symlink that escapes is refused
440
+ * even when it points somewhere "useful" (e.g. a node_modules symlinked to a
441
+ * sibling checkout) — the model can still reach such paths through
442
+ * execute_command, which has its own permission gate.
443
+ */
444
+ export function resolveWithinWorkspace(root: string, p: string): string {
445
+ const rootAbs = path.resolve(root)
446
+ const target = path.resolve(rootAbs, p)
447
+ if (target !== rootAbs && !target.startsWith(rootAbs + path.sep)) {
448
+ throw new PathTraversalError(p, rootAbs)
449
+ }
450
+ assertNoSymlinkEscape(rootAbs, target, p)
451
+ return target
452
+ }
453
+
454
+ /**
455
+ * Reject a `target` whose real location (after following symlinks) escapes
456
+ * `rootAbs`. Called by resolveWithinWorkspace after the lexical check.
457
+ *
458
+ * Walk up from `target` until an existing path is found, realpath it, and
459
+ * require the result to stay inside the root's own realpath. Three cases
460
+ * realpath can't answer directly, each handled explicitly:
461
+ * - Path doesn't exist yet (a write to a new file): walk up to the deepest
462
+ * existing ancestor — its real location decides where the write lands.
463
+ * - A component is a DANGLING symlink: realpath fails, but a write through
464
+ * it would create the target AT THE SYMLINK'S DESTINATION, so follow the
465
+ * chain (re-running the containment check on each hop) instead of walking
466
+ * past it.
467
+ * - A symlink cycle (realpath throws ELOOP): unresolvable — the kernel
468
+ * refuses reads/writes through it too, so nothing can escape; walk up.
469
+ */
470
+ function assertNoSymlinkEscape(rootAbs: string, target: string, requested: string): void {
471
+ const rootReal = realpathOrSelf(rootAbs)
472
+ const seen = new Set<string>()
473
+ let probe = target
474
+ for (;;) {
475
+ if (seen.has(probe)) {
476
+ // Symlink cycle: unresolvable, so no read/write can escape through
477
+ // it — walk up and keep checking the ancestors.
478
+ const parent = path.dirname(probe)
479
+ if (parent === probe) {
480
+ return
481
+ }
482
+ probe = parent
483
+ continue
484
+ }
485
+ seen.add(probe)
486
+
487
+ let real: string
488
+ try {
489
+ real = fs.realpathSync(probe)
490
+ } catch {
491
+ let st: fs.Stats | undefined
492
+ try {
493
+ st = fs.lstatSync(probe)
494
+ } catch {
495
+ st = undefined
496
+ }
497
+ if (st?.isSymbolicLink()) {
498
+ // Dangling symlink: a later write would create the target at
499
+ // the symlink's destination, so verify the destination chain
500
+ // instead of skipping the symlink.
501
+ probe = path.resolve(path.dirname(probe), fs.readlinkSync(probe))
502
+ continue
503
+ }
504
+ const parent = path.dirname(probe)
505
+ if (parent === probe) {
506
+ return
507
+ }
508
+ probe = parent
509
+ continue
510
+ }
511
+ if (real !== rootReal && !real.startsWith(rootReal + path.sep)) {
512
+ throw new SymlinkEscapeError(requested, rootAbs, real)
513
+ }
514
+ return
515
+ }
516
+ }
517
+
518
+ /** realpath of `p`, falling back to the lexical path when it can't resolve. */
519
+ function realpathOrSelf(p: string): string {
520
+ try {
521
+ return fs.realpathSync(p)
522
+ } catch {
523
+ return p
524
+ }
525
+ }
526
+
527
+ /** Build a path-safety-checked absolute path, catching traversal errors. */
528
+ function safeTarget(ctx: ToolContext, p: string): string {
529
+ return resolveWithinWorkspace(ctx.workspaceRoot, p)
530
+ }
531
+
532
+ // ─── Result helpers ──────────────────────────────────────────────────────────
533
+
534
+ function ok(content: string): ToolResult {
535
+ return { content: truncate(content), isError: false }
536
+ }
537
+
538
+ function err(content: string): ToolResult {
539
+ return { content: truncate(`[Error] ${content}`), isError: true }
540
+ }
541
+
542
+ /**
543
+ * Build the model-facing refusal message for a blocked execute_command. The
544
+ * tone matches the other err(...) messages in this file: name what was
545
+ * refused and why, and give a well-behaved model a concrete way to adjust.
546
+ */
547
+ function refusalMessage(command: string, refusal: CommandRefusal): ToolResult {
548
+ switch (refusal.kind) {
549
+ case "dangerous":
550
+ return err(
551
+ `execute_command: refusing to run '${command}': it contains a dangerous shell substitution pattern ` +
552
+ `(e.g. \${var@P}, \${!var}, <<<\$(...), =(...), or *(e:...:)) which is ALWAYS blocked and cannot be ` +
553
+ `allow-listed or configured away. Rewrite the command without shell parameter-expansion tricks.`,
554
+ )
555
+ case "malformed":
556
+ return err(
557
+ `execute_command: refusing to run '${command}': malformed command (` +
558
+ `${refusal.parseError?.message ?? "shell syntax error"}) — a shell syntax error is never auto-approved. ` +
559
+ `Fix the quoting and retry.`,
560
+ )
561
+ case "denied":
562
+ return err(
563
+ `execute_command: refusing to run '${command}': sub-command '${refusal.subCommand}' is denied by the ` +
564
+ `permissions policy (matches denied pattern '${refusal.pattern}'). Adjust your approach; this command ` +
565
+ `is not permitted even if other parts of the chain are allowed.`,
566
+ )
567
+ case "not_allowed":
568
+ return err(
569
+ `execute_command: refusing to run '${command}': sub-command '${refusal.subCommand}' is not in the ` +
570
+ `allowed-commands list and cannot be auto-approved in this headless session. Add it via ` +
571
+ `--allowed-commands, HEADLESSCODE_ALLOWED_COMMANDS, or <workspaceRoot>/.headlesscode/permissions.json, ` +
572
+ `or adjust your approach.`,
573
+ )
574
+ case "protected_store":
575
+ return err(
576
+ `execute_command: refusing to run '${command}': sub-command '${refusal.subCommand}' is a recursive delete ` +
577
+ `targeting the shared central store at '${refusal.storeRoot}' (resolved target '${refusal.target}'). ` +
578
+ `The central store is protected BY DEFAULT and this cannot be overridden via --allowed-commands or ` +
579
+ `permissions.json — it is shared across every project on this machine, and no single workspace may ` +
580
+ `delete it. Do not attempt to reset it from inside the harness.`,
581
+ )
582
+ case "redirect_escape": {
583
+ const redirect = refusal.redirect
584
+ const shown = redirect !== undefined ? describeRedirect(redirect) : "an output redirect"
585
+ return err(
586
+ `execute_command: refusing to run '${command}': it redirects output outside the workspace root ` +
587
+ `('${shown}') — the resolved target is outside the workspace and cannot be written from a harness ` +
588
+ `session. This is the same boundary every file tool enforces (write_to_file/apply_diff/... reject ` +
589
+ `outside-workspace paths) and cannot be overridden via --allowed-commands or permissions.json. ` +
590
+ `Write scratch files under <workspaceRoot>/.headlesscode/scratch/ instead.`,
591
+ )
592
+ }
593
+ }
594
+ }
595
+
596
+ function truncate(content: string): string {
597
+ if (content.length <= MAX_RESULT_CHARS) {
598
+ return content
599
+ }
600
+ return (
601
+ content.slice(0, MAX_RESULT_CHARS) +
602
+ `\n…[output truncated at ${MAX_RESULT_CHARS} chars to keep context bounded]`
603
+ )
604
+ }
605
+
606
+ /**
607
+ * Apply the tool-result size discipline to raw handler output.
608
+ *
609
+ * With local summarization OFF (default) this is EXACTLY today's behavior:
610
+ * blunt-truncate over `MAX_RESULT_CHARS`. With it ON, a result that would
611
+ * exceed the cap is instead sent to the local model for compression; on ANY
612
+ * summarizer failure we fall back to the same blunt truncation, so an
613
+ * opted-in session with a broken Ollama behaves identically to a non-opted-in
614
+ * one. Small results never reach the summarizer (that would be pure latency
615
+ * and risk for zero benefit).
616
+ *
617
+ * NOTE: deliberately used ONLY for execute_command output (the large,
618
+ * mostly-noisy command-output case). read_file / write_to_file / diff
619
+ * content must stay verbatim — a summarized diff or file body would be a
620
+ * correctness hazard for the cloud model.
621
+ */
622
+ async function summarizeCommandOutput(content: string): Promise<string> {
623
+ if (content.length <= MAX_RESULT_CHARS) {
624
+ return content
625
+ }
626
+ if (!isLocalSummarizationEnabled()) {
627
+ return truncate(content)
628
+ }
629
+ const summarizer = makeSummarizer()
630
+ if (summarizer === undefined) {
631
+ return truncate(content)
632
+ }
633
+ const input = content.length > SUMMARIZER_INPUT_CAP ? content.slice(0, SUMMARIZER_INPUT_CAP) : content
634
+ return summarizeToolResult(input, summarizer, summarizerLogger)
635
+ }
636
+
637
+ /**
638
+ * The one place summarization is actually performed, and the ONLY reason the
639
+ * executor module imports OllamaOutputSummarizer. Wired in the constructor.
640
+ */
641
+ let summarizerLogger: Pick<Logger, "debug" | "warn"> = {
642
+ debug: () => {},
643
+ warn: (message) => process.stderr.write(`[local-summ] ${message}\n`),
644
+ }
645
+
646
+ /**
647
+ * Module-level session summarizer (one per process). Constructed lazily on the
648
+ * first oversized result of an opted-in session; never constructed for a
649
+ * non-opted-in session. Process-level rather than executor-level because the
650
+ * Ollama client is stateless; one process = one summarizer.
651
+ */
652
+ let toolSummarizer: OllamaOutputSummarizer | undefined
653
+
654
+ /** Bind the session logger (used by the summarizer for non-fatal warnings). */
655
+ export function bindSummarizerLogger(logger: Pick<Logger, "debug" | "warn">): void {
656
+ summarizerLogger = logger
657
+ }
658
+
659
+ function requireString(args: Record<string, unknown>, key: string): string {
660
+ const v = args[key]
661
+ if (typeof v !== "string") {
662
+ throw new Error(`Missing or invalid string argument '${key}' for tool`)
663
+ }
664
+ return v
665
+ }
666
+
667
+ function toNonNegativeInt(v: unknown, fallback: number): number {
668
+ if (typeof v === "number" && Number.isFinite(v)) {
669
+ return Math.max(0, Math.floor(v))
670
+ }
671
+ if (typeof v === "string" && v.trim() !== "" && Number.isFinite(Number(v))) {
672
+ return Math.max(0, Math.floor(Number(v)))
673
+ }
674
+ return fallback
675
+ }
676
+
677
+ // ─── read_file session cache ─────────────────────────────────────────────────
678
+
679
+ /**
680
+ * Session-scoped read_file cache.
681
+ *
682
+ * read_file is the one tool with measured, real waste: a real session re-read
683
+ * the same path with identical args 14-16 times, re-sending byte-identical
684
+ * content at full output-token cost every time (see the DEFAULT_WINDOW_SIZE
685
+ * story in src/engine/loop.ts). Even with that history bug fixed, a model will
686
+ * legitimately re-read a file it saw earlier in a long session — there is no
687
+ * reason to pay full output tokens for content the conversation already has.
688
+ *
689
+ * Correctness: a cache hit requires BOTH (a) the exact effective args that
690
+ * produce byte-identical output, AND (b) the current on-disk content hashing
691
+ * identically to the prior read. (b) is checked by hashing the file at
692
+ * cache-check time — never by tracking "did a write-shaped tool get called",
693
+ * because a file can change for reasons the executor doesn't directly control
694
+ * (execute_command running a formatter/build/codegen, an external editor,
695
+ * anything else). The hash is cheap (node:crypto sha256, no new dependency).
696
+ *
697
+ * The hit short-circuit applies once per "unchanged streak": the first
698
+ * identical call after real content was served returns the short cache-hit
699
+ * message, the SECOND consecutive identical call serves real content again, so
700
+ * a model that is confused or insistent is never stuck being told "it's
701
+ * cached" with no way to actually get the content back.
702
+ *
703
+ * Scope: per ToolExecutor instance, i.e. per session (executors are
704
+ * constructed per-session — see createHeadlessExecutor and its read-only
705
+ * siblings). Deliberately NOT persisted to disk and NOT shared across
706
+ * instances: a reviewer/QA executor gets its own independent cache.
707
+ */
708
+
709
+ /**
710
+ * Everything about a read_file call that affects its output, serialized into a
711
+ * stable string so two calls that produce byte-identical output always collide
712
+ * (see makeReadFileCacheKey).
713
+ */
714
+ type ReadFileCacheKey = {
715
+ /** Resolved absolute path (path.resolve'd, so ./a.ts and a.ts collide). */
716
+ target: string
717
+ /** 'slice' or 'indentation'. */
718
+ mode: string
719
+ /** Effective 1-based offset (default 1). */
720
+ offset: number
721
+ /** Effective limit (default readLimitFromEnv()). */
722
+ limit: number
723
+ /** Effective indentation-mode max_lines (default 60); undefined in slice mode. */
724
+ windowLines?: number
725
+ }
726
+
727
+ /** One cache entry: content identity (hash + size + mtime) + streak state. */
728
+ type ReadFileCacheEntry = {
729
+ /** sha256 of the file content as of the last real read of this key. */
730
+ hash: string
731
+ /** File size at the last real read — half of the fast-path identity. */
732
+ size: number
733
+ /**
734
+ * mtime at the last real read — the other half of the fast-path identity.
735
+ * mtimeMs (float, sub-ms precision) is the highest-resolution mtime this
736
+ * Node exposes on a non-bigint stat (mtimeNs needs { bigint: true }).
737
+ */
738
+ mtimeMs: number
739
+ /**
740
+ * Whether the last read of this key was already served as a cache-hit
741
+ * message. When true, the next identical call serves real content again
742
+ * (resetting this flag), so the hit message never loops forever.
743
+ */
744
+ toldUnchanged: boolean
745
+ }
746
+
747
+ /**
748
+ * Serialize a read_file cache key to a stable string. Map keys must be
749
+ * primitives (object keys compare by identity, so two structurally-identical
750
+ * fresh objects would never collide); a JSON string of the fully-resolved
751
+ * effective args is both stable and collision-free.
752
+ */
753
+ function makeReadFileCacheKey(key: ReadFileCacheKey): string {
754
+ return JSON.stringify(key)
755
+ }
756
+
757
+ /** Cache-hit message shown instead of the full file content. */
758
+ const READ_FILE_CACHE_HIT_MESSAGE =
759
+ "[cache] this file is unchanged since your last read of it earlier in this session (identical content, same range). Re-read the earlier tool result for the content, or call read_file again if you specifically need it re-sent."
760
+
761
+ function hashFileContent(content: string): string {
762
+ return createHash("sha256").update(content).digest("hex")
763
+ }
764
+
765
+ /** The indentation-mode window size this harness actually uses (Phase 1 minimal). */
766
+ const DEFAULT_INDENTATION_WINDOW_LINES = 60
767
+
768
+ /** read_file — slice mode with offset/limit pagination (offset is 1-based). */
769
+ function readFileHandler(
770
+ args: Record<string, unknown>,
771
+ ctx: ToolContext,
772
+ cache: ReadFileCacheEntry | undefined,
773
+ entry: Map<string, ReadFileCacheEntry>,
774
+ ): Promise<ToolResult> {
775
+ const filePath = requireString(args, "path")
776
+ return (async () => {
777
+ const target = safeTarget(ctx, filePath)
778
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
779
+
780
+ let stat: fs.Stats
781
+ try {
782
+ stat = await fsp.stat(target)
783
+ } catch (error) {
784
+ return err(`read_file: cannot stat '${rel}': ${errorMessage(error)}`)
785
+ }
786
+ if (!stat.isFile()) {
787
+ return err(`read_file: '${rel}' is not a file`)
788
+ }
789
+
790
+ const mode = typeof args.mode === "string" ? args.mode : "slice"
791
+ const offset = toNonNegativeInt(args.offset, 1) // 1-based
792
+ const limit = toNonNegativeInt(args.limit, readLimitFromEnv())
793
+ const indentation = args.indentation as Record<string, unknown> | undefined
794
+ const windowLines = mode === "indentation" ? toNonNegativeInt(indentation?.["max_lines"], DEFAULT_INDENTATION_WINDOW_LINES) : undefined
795
+
796
+ const key = makeReadFileCacheKey({ target, mode, offset, limit, windowLines })
797
+
798
+ // Trust model: unchanged size AND mtime ⇒ identical content, so the
799
+ // cached hash can be reused without re-reading the file; any mismatch
800
+ // (including a same-length rewrite, which changes mtime) falls back to
801
+ // a full read + sha256 below.
802
+ if (
803
+ cache !== undefined &&
804
+ !cache.toldUnchanged &&
805
+ cache.size === stat.size &&
806
+ cache.mtimeMs === stat.mtimeMs
807
+ ) {
808
+ cache.toldUnchanged = true
809
+ return ok(READ_FILE_CACHE_HIT_MESSAGE)
810
+ }
811
+
812
+ let content: string
813
+ try {
814
+ content = await fsp.readFile(target, "utf-8")
815
+ } catch (error) {
816
+ return err(`read_file: cannot read '${rel}': ${errorMessage(error)}`)
817
+ }
818
+
819
+ // Cache-check the CURRENT on-disk content (never "no write tool was
820
+ // called"): identical args + identical hash => byte-identical output.
821
+ const currentHash = hashFileContent(content)
822
+ if (cache !== undefined && cache.hash === currentHash && !cache.toldUnchanged) {
823
+ cache.toldUnchanged = true
824
+ return ok(READ_FILE_CACHE_HIT_MESSAGE)
825
+ }
826
+
827
+ const allLines = content.split(/\r?\n/)
828
+ let result: ToolResult
829
+ if (mode === "indentation") {
830
+ // Phase 1 minimal: indentation mode falls back to a window around the
831
+ // anchor line (anchor_line 1-based), which is good enough for the loop.
832
+ const anchor = toNonNegativeInt(indentation?.["anchor_line"], offset)
833
+ result = ok(formatFileSlice(rel, allLines, Math.max(1, anchor), windowLines ?? DEFAULT_INDENTATION_WINDOW_LINES))
834
+ } else {
835
+ result = ok(formatFileSlice(rel, allLines, Math.max(1, offset), limit))
836
+ }
837
+
838
+ // Serve (or re-serve) real content; record identity + reset the hit
839
+ // flag so the next identical call may short-circuit once more.
840
+ entry.set(key, { hash: currentHash, size: stat.size, mtimeMs: stat.mtimeMs, toldUnchanged: false })
841
+ return result
842
+ })()
843
+ }
844
+
845
+ /**
846
+ * Compute the stable string cache key for a read_file call, mirroring exactly
847
+ * how readFileHandler resolves its args (defaults and all) so two calls that
848
+ * produce byte-identical output always collide on the same key. Path safety is
849
+ * enforced identically to the handler itself, so an escaping path errors here
850
+ * exactly as it would in the handler (and caches nothing).
851
+ */
852
+ function readFileKey(args: Record<string, unknown>, ctx: ToolContext): string {
853
+ const filePath = requireString(args, "path")
854
+ const target = safeTarget(ctx, filePath)
855
+ const mode = typeof args.mode === "string" ? args.mode : "slice"
856
+ const offset = toNonNegativeInt(args.offset, 1) // 1-based
857
+ const limit = toNonNegativeInt(args.limit, readLimitFromEnv())
858
+ const indentation = args.indentation as Record<string, unknown> | undefined
859
+ const windowLines =
860
+ mode === "indentation" ? toNonNegativeInt(indentation?.["max_lines"], DEFAULT_INDENTATION_WINDOW_LINES) : undefined
861
+ return makeReadFileCacheKey({ target, mode, offset, limit, windowLines })
862
+ }
863
+
864
+ function formatFileSlice(rel: string, allLines: string[], offset: number, limit: number): string {
865
+ const start = Math.max(1, offset)
866
+ const slice = allLines.slice(start - 1, start - 1 + limit)
867
+ const totalLines = allLines.length
868
+ const body = slice.map((line, i) => `${start + i} | ${line}`).join("\n")
869
+
870
+ const header = `File: ${rel}`
871
+ if (totalLines > start - 1 + limit) {
872
+ return `${header}\nShowing lines ${start}-${start + slice.length - 1} of ${totalLines} total lines (use read_file with offset=${start + limit} to read more).\n${body}`
873
+ }
874
+ return `${header}\n${body}`.replace(/\n$/, "")
875
+ }
876
+
877
+ /**
878
+ * Shared protected-files guard for every write tool. Returns a refusal
879
+ * ToolResult when `rel` (workspace-relative, POSIX-separated) matches a
880
+ * protected pattern and the escape hatch — --allow-protected-writes or
881
+ * "allowProtectedWrites": true in .headlesscode/permissions.json — is
882
+ * explicitly on (OFF by default). Naming the matched pattern gives a
883
+ * well-behaved model a concrete reason to stop. A refusal is a real tool
884
+ * error (isError: true) and counts toward the consecutive-mistake bound,
885
+ * exactly like any other tool failure.
886
+ */
887
+ function protectedWriteRefusal(toolName: string, rel: string, permissions: PermissionsConfig): ToolResult | null {
888
+ if (permissions.allowProtectedWrites) {
889
+ return null
890
+ }
891
+ const matchedPattern = findMatchingPattern(rel, permissions.protectedFiles)
892
+ if (matchedPattern === null) {
893
+ return null
894
+ }
895
+ return err(
896
+ `${toolName}: refusing to write protected file '${rel}' (matches protected pattern '${matchedPattern}'). ` +
897
+ `This file is protected by the harness permissions policy and cannot be overwritten. If this write is ` +
898
+ `genuinely required, re-run with --allow-protected-writes (or set "allowProtectedWrites": true in ` +
899
+ `<workspaceRoot>/.headlesscode/permissions.json); it is OFF by default.`,
900
+ )
901
+ }
902
+
903
+ /**
904
+ * Existing-file content length (bytes) above which write_to_file refuses to
905
+ * overwrite when ctx.guardLargeOverwrites is on. Chosen well above trivial
906
+ * stub/placeholder content (empty scaffolds, one-liners) so the common
907
+ * legitimate case — write_to_file creating or replacing a small/new file —
908
+ * is never affected; see largeOverwriteRefusal's doc comment.
909
+ */
910
+ const LARGE_OVERWRITE_GUARD_BYTES = 200
911
+
912
+ /**
913
+ * guardLargeOverwrites (see ToolContext.guardLargeOverwrites, types.ts):
914
+ * refuse write_to_file against a file that already exists and has
915
+ * substantial content, mirroring edit_file's empty-old_string refusal in
916
+ * the other direction. Verified live 2026-08-20 against Qwen2.5-Coder-14B
917
+ * and Qwen3-14B: given a real ~500-line file and a one-function-add task,
918
+ * both had a strong bias toward regenerating the ENTIRE file from scratch
919
+ * via write_to_file instead of a targeted diff — and since a full
920
+ * regeneration needs far more output budget than a precise edit, this
921
+ * reliably truncates mid-file, silently destroying everything after the
922
+ * cutoff. The escape hatch (delete-then-write) is deliberate: it requires a
923
+ * SEPARATE, explicit destructive action instead of one accidental call, so
924
+ * a genuine full-file rewrite is still possible without disabling the
925
+ * guard.
926
+ */
927
+ async function largeOverwriteRefusal(target: string, rel: string, ctx: ToolContext): Promise<ToolResult | null> {
928
+ if (!ctx.guardLargeOverwrites) {
929
+ return null
930
+ }
931
+ let existingSize: number
932
+ try {
933
+ existingSize = (await fsp.stat(target)).size
934
+ } catch {
935
+ return null // Target doesn't exist yet — the legitimate new-file case.
936
+ }
937
+ if (existingSize <= LARGE_OVERWRITE_GUARD_BYTES) {
938
+ return null
939
+ }
940
+ return err(
941
+ `write_to_file: refusing to overwrite '${rel}' (${existingSize} bytes of existing content).\n\n` +
942
+ `<error_details>\nwrite_to_file replaces this file's ENTIRE content. For an existing file of this size, ` +
943
+ `regenerating it from scratch instead of making a targeted change risks silently losing content that ` +
944
+ `isn't reproduced (especially if generation is cut off before finishing the full file).\n\n` +
945
+ `Recovery suggestions:\n1. Use edit_file or search_replace to make a precise, targeted change instead\n` +
946
+ `2. Use read_file first if you haven't seen the file's current contents\n3. If a full-file rewrite is ` +
947
+ `genuinely intended, delete the file first (execute_command) — write_to_file always succeeds against a ` +
948
+ `path that doesn't exist\n</error_details>`,
949
+ )
950
+ }
951
+
952
+ /** write_to_file — create parent dirs as needed, overwrite existing files. */
953
+ function writeToFileHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
954
+ const filePath = requireString(args, "path")
955
+ const content = requireString(args, "content")
956
+ return (async () => {
957
+ const target = safeTarget(ctx, filePath)
958
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
959
+
960
+ // Permissions: refuse writes to protected files (secret/credential
961
+ // patterns) unless the escape hatch — --allow-protected-writes or
962
+ // "allowProtectedWrites": true in .headlesscode/permissions.json — is
963
+ // explicitly on (OFF by default). All four write tools share the
964
+ // protectedWriteRefusal helper below, so no write path can bypass it.
965
+ const refusal = protectedWriteRefusal("write_to_file", rel, ctx.permissions)
966
+ if (refusal !== null) {
967
+ return refusal
968
+ }
969
+
970
+ const overwriteRefusal = await largeOverwriteRefusal(target, rel, ctx)
971
+ if (overwriteRefusal !== null) {
972
+ return overwriteRefusal
973
+ }
974
+
975
+ try {
976
+ await fsp.mkdir(path.dirname(target), { recursive: true })
977
+ await fsp.writeFile(target, content, "utf-8")
978
+ } catch (error) {
979
+ return err(`write_to_file: failed to write '${rel}': ${errorMessage(error)}`)
980
+ }
981
+ return ok(`File written: ${rel} (${Buffer.byteLength(content, "utf-8")} bytes)`)
982
+ })()
983
+ }
984
+
985
+ // ─── execute_command backgrounded-child registry ─────────────────────────────
986
+
987
+ /**
988
+ * Children that timed out and were intentionally left running in the
989
+ * background (see executeCommandHandler). Tracked for exactly two reasons:
990
+ * (a) their stdout/stderr pipes keep being drained so a long-running child
991
+ * never blocks on a full pipe buffer, and
992
+ * (b) `ToolExecutor.dispose()` can hard-kill anything still running when a
993
+ * session truly ends, so the harness never orphans a process.
994
+ * A later `execute_command` mid-session (e.g. `pkill`, `docker compose down`)
995
+ * works unchanged — the model targets the process by port/name/pattern, same
996
+ * as a human would.
997
+ */
998
+ const backgroundCommands = new Set<ChildProcess>()
999
+
1000
+ /** Best-effort unref of a stdio pipe so it can't keep the event loop alive. */
1001
+ function unrefStream(stream: NodeJS.ReadableStream | null): void {
1002
+ try {
1003
+ // child.stdout/stderr are net.Socket instances at runtime but typed as
1004
+ // Readable, so feature-detect `unref` rather than casting to Socket.
1005
+ ;(stream as { unref?: () => void } | null)?.unref?.()
1006
+ } catch {
1007
+ // Not every stream type supports unref — the child's own unref (see
1008
+ // executeCommandHandler) is the important part for harness exit.
1009
+ }
1010
+ }
1011
+
1012
+ /**
1013
+ * Session teardown: hard-kill every execute_command child still running after
1014
+ * a timeout. Called by `ToolExecutor.dispose()` when a session ends (success,
1015
+ * bounded failure, budget abort, or thrown error) so no backgrounded process
1016
+ * is orphaned past its session. Deliberately NOT called when a single tool
1017
+ * call times out mid-session — that is the whole point of the background
1018
+ * semantics (see the vendored execute_command tool's timeout contract).
1019
+ */
1020
+ function killBackgroundCommands(): void {
1021
+ for (const child of backgroundCommands) {
1022
+ try {
1023
+ if (child.exitCode === null && child.signalCode === null) {
1024
+ if (child.pid != null) {
1025
+ // The child is a detached process-group leader (see
1026
+ // executeCommandHandler), so kill the whole group with a
1027
+ // negative pid — that reaps grandchildren too (e.g. the
1028
+ // `node`/`sh` the shell may have spawned), not just the
1029
+ // shell itself. Falls back to killing the direct child on
1030
+ // platforms where group signals aren't supported.
1031
+ try {
1032
+ process.kill(-child.pid, "SIGKILL")
1033
+ } catch {
1034
+ child.kill("SIGKILL")
1035
+ }
1036
+ } else {
1037
+ child.kill("SIGKILL")
1038
+ }
1039
+ }
1040
+ } catch {
1041
+ // Already gone — nothing to clean up.
1042
+ }
1043
+ }
1044
+ backgroundCommands.clear()
1045
+ }
1046
+
1047
+ // ─── apply_diff — vendored MultiSearchReplaceDiffStrategy ────────────────────
1048
+
1049
+ /**
1050
+ * apply_diff — surgical edits from one or more SEARCH/REPLACE blocks in a
1051
+ * single `diff` string. Mirrors the upstream ApplyDiffTool flow, adapted to
1052
+ * this project's ToolResult/ok/err conventions: the diff string is fed to the
1053
+ * vendored MultiSearchReplaceDiffStrategy (fuzzy Levenshtein matching + optional
1054
+ * `:start_line:` disambiguation), and on success the merged content is written
1055
+ * straight to disk.
1056
+ */
1057
+ async function applyDiffHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
1058
+ const filePath = requireString(args, "path")
1059
+ const diffContent = requireString(args, "diff")
1060
+ return (async () => {
1061
+ const target = safeTarget(ctx, filePath)
1062
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
1063
+
1064
+ // Permissions: same protected-files guard as write_to_file (shared
1065
+ // helper) — an agent must not edit .env / *.pem / *.key through a
1066
+ // different write tool than the one the guard was originally wired to.
1067
+ const refusal = protectedWriteRefusal("apply_diff", rel, ctx.permissions)
1068
+ if (refusal !== null) {
1069
+ return refusal
1070
+ }
1071
+
1072
+ let originalContent: string
1073
+ try {
1074
+ originalContent = await fsp.readFile(target, "utf-8")
1075
+ } catch (error) {
1076
+ return err(
1077
+ `apply_diff: file does not exist at '${rel}' (or could not be read): ${errorMessage(
1078
+ error,
1079
+ )}\n\nUse write_to_file to create new files; apply_diff edits existing files only.`,
1080
+ )
1081
+ }
1082
+
1083
+ const strategy = new MultiSearchReplaceDiffStrategy()
1084
+ const diffResult = await strategy.applyDiff(originalContent, diffContent)
1085
+
1086
+ if (!diffResult.success) {
1087
+ // Surface the first failing part's error (most actionable), else the
1088
+ // strategy-level error.
1089
+ const failPart = diffResult.failParts?.find((p) => !p.success)
1090
+ const detail = failPart?.error ?? diffResult.error ?? "Unknown diff error"
1091
+ return err(`apply_diff: unable to apply diff to '${rel}':\n\n${detail}`)
1092
+ }
1093
+
1094
+ // Write the merged content back to disk.
1095
+ try {
1096
+ await fsp.writeFile(target, diffResult.content, "utf-8")
1097
+ } catch (error) {
1098
+ return err(`apply_diff: failed to write '${rel}': ${errorMessage(error)}`)
1099
+ }
1100
+
1101
+ const failedParts = (diffResult.failParts ?? []).filter((p) => !p.success)
1102
+ let message = `File updated: ${rel}`
1103
+ if (failedParts.length > 0) {
1104
+ message += `\nBut unable to apply all diff parts to file: ${rel} (${failedParts.length} failed). Use the read_file tool to check the newest file version and re-apply diffs.`
1105
+ }
1106
+ // Single SEARCH/REPLACE block notice (mirrors ApplyDiffTool). The marker
1107
+ // literal is split to avoid confusing the harness's own diff parser.
1108
+ const searchMarker = "<<<<<<<" + " SEARCH"
1109
+ const searchBlocks = diffContent.split(searchMarker).length - 1
1110
+ if (searchBlocks === 1) {
1111
+ message +=
1112
+ "\n<notice>Making multiple related changes in a single apply_diff is more efficient. If other changes are needed in this file, please include them as additional SEARCH/REPLACE blocks.</notice>"
1113
+ }
1114
+ return ok(message)
1115
+ })()
1116
+ }
1117
+
1118
+ // ─── search_replace — strict literal one-occurrence replacement ──────────────
1119
+
1120
+ /**
1121
+ * search_replace — a literal string replacement requiring old_string to match
1122
+ * EXACTLY once (the core safety property: never guess which occurrence the
1123
+ * model meant). Normalizes line endings to LF for matching, mirroring the
1124
+ * upstream SearchReplaceTool.
1125
+ */
1126
+ async function searchReplaceHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
1127
+ const filePath = requireString(args, "file_path")
1128
+ const oldString = requireString(args, "old_string")
1129
+ const newString = requireString(args, "new_string")
1130
+ return (async () => {
1131
+ // Upstream fails on empty old_string (treated as a missing parameter) —
1132
+ // it must never be a silent no-op or a split("") character count.
1133
+ if (oldString === "") {
1134
+ return err("search_replace: missing 'old_string' — it must be a non-empty string to search for.")
1135
+ }
1136
+ if (oldString === newString) {
1137
+ return err("search_replace: 'old_string' and 'new_string' must be different.")
1138
+ }
1139
+
1140
+ const target = safeTarget(ctx, filePath)
1141
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
1142
+
1143
+ // Permissions: same protected-files guard as write_to_file (shared
1144
+ // helper) — an agent must not edit .env / *.pem / *.key through a
1145
+ // different write tool than the one the guard was originally wired to.
1146
+ const refusal = protectedWriteRefusal("search_replace", rel, ctx.permissions)
1147
+ if (refusal !== null) {
1148
+ return refusal
1149
+ }
1150
+
1151
+ let fileContent: string
1152
+ try {
1153
+ fileContent = await fsp.readFile(target, "utf-8")
1154
+ } catch (error) {
1155
+ return err(
1156
+ `search_replace: file not found at '${rel}' (or could not be read): ${errorMessage(
1157
+ error,
1158
+ )}\n\nCannot perform search and replace on a non-existent file.`,
1159
+ )
1160
+ }
1161
+
1162
+ // Normalize line endings to LF for consistent matching (upstream behavior).
1163
+ fileContent = fileContent.replace(/\r\n/g, "\n")
1164
+ const normalizedOld = oldString.replace(/\r\n/g, "\n")
1165
+ const normalizedNew = newString.replace(/\r\n/g, "\n")
1166
+
1167
+ const matchCount = fileContent.split(normalizedOld).length - 1
1168
+
1169
+ if (matchCount === 0) {
1170
+ return err(
1171
+ `search_replace: no match found for 'old_string' in '${rel}'. Please ensure it matches the file contents exactly, including whitespace and indentation.`,
1172
+ )
1173
+ }
1174
+ if (matchCount > 1) {
1175
+ return err(
1176
+ `search_replace: found ${matchCount} matches for 'old_string' in '${rel}'. This tool can only replace ONE occurrence at a time. Please provide more context (3-5 lines before and after) to uniquely identify the specific instance you want to change.`,
1177
+ )
1178
+ }
1179
+
1180
+ const newContent = fileContent.replace(normalizedOld, normalizedNew)
1181
+ if (newContent === fileContent) {
1182
+ return ok(`No changes needed for '${rel}'`)
1183
+ }
1184
+
1185
+ try {
1186
+ await fsp.writeFile(target, newContent, "utf-8")
1187
+ } catch (error) {
1188
+ return err(`search_replace: failed to write '${rel}': ${errorMessage(error)}`)
1189
+ }
1190
+ return ok(`File updated: ${rel}`)
1191
+ })()
1192
+ }
1193
+
1194
+ // ─── edit_file — fallback matching chain + file creation ─────────────────────
1195
+
1196
+ type LineEnding = "\r\n" | "\n"
1197
+
1198
+ /**
1199
+ * Character-count growth (new_string longer than old_string) above which a
1200
+ * multi-site edit_file replacement (expected_replacements > 1) is refused.
1201
+ * See the guard's call site in editFileHandler for the live-verified
1202
+ * failure this exists to prevent.
1203
+ */
1204
+ const UNSAFE_MULTI_REPLACE_GROWTH_CHARS = 40
1205
+
1206
+ /**
1207
+ * Count occurrences of a substring in a string (non-overlapping).
1208
+ * Ported verbatim from upstream EditFileTool.ts.
1209
+ */
1210
+ function countOccurrences(str: string, substr: string): number {
1211
+ if (substr === "") return 0
1212
+ let count = 0
1213
+ let pos = str.indexOf(substr)
1214
+ while (pos !== -1) {
1215
+ count++
1216
+ pos = str.indexOf(substr, pos + substr.length)
1217
+ }
1218
+ return count
1219
+ }
1220
+
1221
+ /**
1222
+ * Safely replace all occurrences of a literal string, handling $ escape
1223
+ * sequences. Ported verbatim from upstream EditFileTool.ts.
1224
+ */
1225
+ function safeLiteralReplace(str: string, oldString: string, newString: string): string {
1226
+ if (oldString === "" || !str.includes(oldString)) {
1227
+ return str
1228
+ }
1229
+ if (!newString.includes("$")) {
1230
+ return str.replaceAll(oldString, newString)
1231
+ }
1232
+ const escapedNewString = newString.replaceAll("$", "$$$$")
1233
+ return str.replaceAll(oldString, escapedNewString)
1234
+ }
1235
+
1236
+ function detectLineEnding(content: string): LineEnding {
1237
+ return content.includes("\r\n") ? "\r\n" : "\n"
1238
+ }
1239
+
1240
+ function normalizeToLF(content: string): string {
1241
+ return content.replace(/\r\n/g, "\n")
1242
+ }
1243
+
1244
+ function restoreLineEnding(contentLF: string, eol: LineEnding): string {
1245
+ if (eol === "\n") return contentLF
1246
+ return contentLF.replace(/\n/g, "\r\n")
1247
+ }
1248
+
1249
+ function escapeRegExp(input: string): string {
1250
+ return input.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")
1251
+ }
1252
+
1253
+ /**
1254
+ * Whitespace-tolerant regex: treats runs of horizontal whitespace and
1255
+ * cross-line whitespace as flexible, so minor formatting drift still matches.
1256
+ * Ported verbatim from upstream EditFileTool.ts.
1257
+ */
1258
+ function buildWhitespaceTolerantRegex(oldLF: string): RegExp {
1259
+ if (oldLF === "") {
1260
+ return new RegExp("(?!)", "g")
1261
+ }
1262
+
1263
+ const parts = oldLF.match(/(\s+|\S+)/g) ?? []
1264
+ const whitespacePatternForRun = (run: string): string => {
1265
+ if (run.includes("\n")) {
1266
+ return "\\s+"
1267
+ }
1268
+ return "[\\t ]+"
1269
+ }
1270
+
1271
+ const pattern = parts
1272
+ .map((part) => {
1273
+ if (/^\s+$/.test(part)) {
1274
+ return whitespacePatternForRun(part)
1275
+ }
1276
+ return escapeRegExp(part)
1277
+ })
1278
+ .join("")
1279
+
1280
+ return new RegExp(pattern, "g")
1281
+ }
1282
+
1283
+ /**
1284
+ * Token-based regex: matches the non-whitespace tokens in order, separated by
1285
+ * any whitespace. Ported verbatim from upstream EditFileTool.ts.
1286
+ */
1287
+ function buildTokenRegex(oldLF: string): RegExp {
1288
+ const tokens = oldLF.split(/\s+/).filter(Boolean)
1289
+ if (tokens.length === 0) {
1290
+ return new RegExp("(?!)", "g")
1291
+ }
1292
+
1293
+ const pattern = tokens.map(escapeRegExp).join("\\s+")
1294
+ return new RegExp(pattern, "g")
1295
+ }
1296
+
1297
+ function countRegexMatches(content: string, regex: RegExp): number {
1298
+ const stable = new RegExp(regex.source, regex.flags)
1299
+ return Array.from(content.matchAll(stable)).length
1300
+ }
1301
+
1302
+ /**
1303
+ * edit_file — literal string replacement resilient to formatting drift via the
1304
+ * fallback chain exact → whitespace-tolerant → token-based, with an optional
1305
+ * `expected_replacements` count (default 1). Also creates new files when
1306
+ * old_string is "" (failing clearly if the file already exists). Mirrors the
1307
+ * upstream EditFileTool; the file's original line endings are preserved.
1308
+ */
1309
+ async function editFileHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
1310
+ // Coerce old_string/new_string to handle malformed native calls that pass
1311
+ // non-strings (upstream normalizes those to "" to avoid later crashes).
1312
+ const filePath = requireString(args, "file_path")
1313
+ const oldString = typeof args.old_string === "string" ? args.old_string : ""
1314
+ const newString = typeof args.new_string === "string" ? args.new_string : ""
1315
+ const expectedReplacements = Math.max(1, toNonNegativeInt(args.expected_replacements, 1))
1316
+
1317
+ return (async () => {
1318
+ const target = safeTarget(ctx, filePath)
1319
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
1320
+
1321
+ // Permissions: same protected-files guard as write_to_file (shared
1322
+ // helper) — an agent must not edit .env / *.pem / *.key through a
1323
+ // different write tool than the one the guard was originally wired to.
1324
+ const refusal = protectedWriteRefusal("edit_file", rel, ctx.permissions)
1325
+ if (refusal !== null) {
1326
+ return refusal
1327
+ }
1328
+
1329
+ let currentContent: string | null = null
1330
+ let currentContentLF: string | null = null
1331
+ let originalEol: LineEnding = "\n"
1332
+ let isNewFile = false
1333
+
1334
+ // Read the file (or determine it doesn't exist / we're creating it).
1335
+ let fileExists = false
1336
+ try {
1337
+ await fsp.access(target)
1338
+ fileExists = true
1339
+ } catch {
1340
+ fileExists = false
1341
+ }
1342
+
1343
+ if (fileExists) {
1344
+ try {
1345
+ currentContent = await fsp.readFile(target, "utf8")
1346
+ originalEol = detectLineEnding(currentContent)
1347
+ currentContentLF = normalizeToLF(currentContent)
1348
+ } catch (error) {
1349
+ return err(
1350
+ `edit_file: failed to read file '${rel}': ${errorMessage(
1351
+ error,
1352
+ )}\n\nRecovery suggestions:\n1. Verify the file exists and is readable\n2. Check file permissions\n3. If the file may have changed, use read_file to confirm its current contents`,
1353
+ )
1354
+ }
1355
+
1356
+ // Check if trying to create a file that already exists.
1357
+ if (oldString === "") {
1358
+ return err(
1359
+ `edit_file: file already exists: '${rel}'\n\n<error_details>\nYou provided an empty old_string, which indicates file creation, but the target file already exists.\n\nRecovery suggestions:\n1. To modify an existing file, provide a non-empty old_string that matches the current file contents\n2. Use read_file to confirm the exact text to match\n3. If you intended to overwrite the entire file, use write_to_file instead\n</error_details>`,
1360
+ )
1361
+ }
1362
+ } else {
1363
+ if (oldString === "") {
1364
+ // Creating a new file.
1365
+ isNewFile = true
1366
+ } else {
1367
+ return err(
1368
+ `edit_file: file does not exist at path: '${rel}'\n\n<error_details>\nThe specified file could not be found, so the replacement could not be performed.\n\nRecovery suggestions:\n1. Verify the file path is correct\n2. If you intended to create a new file, set old_string to an empty string\n3. Use list_files or read_file to confirm the correct path\n</error_details>`,
1369
+ )
1370
+ }
1371
+ }
1372
+
1373
+ const oldLF = normalizeToLF(oldString)
1374
+ const newLF = normalizeToLF(newString)
1375
+
1376
+ // Validate the replacement operation on an existing file.
1377
+ if (!isNewFile && currentContentLF !== null) {
1378
+ if (oldLF === newLF) {
1379
+ return err(
1380
+ `edit_file: no changes to apply for file '${rel}'\n\n<error_details>\nThe provided old_string and new_string are identical (after normalizing line endings), so there is nothing to change.\n\nRecovery suggestions:\n1. Update new_string to the intended replacement text\n2. If you intended to verify file state only, use read_file instead\n</error_details>`,
1381
+ )
1382
+ }
1383
+
1384
+ // Unsafe-multi-replace-insertion guard: verified live 2026-08-20 —
1385
+ // a local model's old_string ("estimateMessageChars", a bare
1386
+ // identifier) matched 4 unrelated sites (one function definition,
1387
+ // three call sites) in condense.ts. edit_file's own error message
1388
+ // on the first (correctly refused, count-mismatch) attempt
1389
+ // suggested "if you intend to replace all occurrences, set
1390
+ // expected_replacements to N" — reasonable advice for a genuine
1391
+ // rename, but the model's real intent was to INSERT a large new
1392
+ // function body once, right after the definition. Taking that
1393
+ // suggestion literally applied the same large insertion at all 4
1394
+ // sites, corrupting the 3 call sites (each ended up with the new
1395
+ // function body spliced into the middle of a function call). A
1396
+ // multi-site replacement whose new_string is much LONGER than
1397
+ // old_string is exactly the insertion shape, not the rename
1398
+ // shape (a genuine rename keeps old_string and new_string close
1399
+ // in length) — refuse it up front rather than let the "set
1400
+ // expected_replacements" suggestion above walk a model into this.
1401
+ if (expectedReplacements > 1 && newLF.length - oldLF.length > UNSAFE_MULTI_REPLACE_GROWTH_CHARS) {
1402
+ return err(
1403
+ `edit_file: refusing expected_replacements=${expectedReplacements} — new_string is ${newLF.length - oldLF.length} characters longer than old_string.\n\n` +
1404
+ `<error_details>\nReplacing several sites at once with a much LARGER block of text is almost always a mistake: ` +
1405
+ `it means the SAME large insertion would be spliced into every matching location, not just the one you actually ` +
1406
+ `intend to change. A genuine multi-site replacement (a rename, for example) keeps old_string and new_string close ` +
1407
+ `in length.\n\nRecovery suggestions:\n1. Use read_file to see the exact surrounding context, then include enough of ` +
1408
+ `it in old_string to uniquely identify the ONE location you actually want to change\n2. Set expected_replacements back ` +
1409
+ `to 1 once old_string is unique\n3. If you genuinely want the SAME large content at multiple locations, make separate ` +
1410
+ `edit_file calls, one per location, each with a uniquely-identifying old_string\n</error_details>`,
1411
+ )
1412
+ }
1413
+
1414
+ const wsRegex = buildWhitespaceTolerantRegex(oldLF)
1415
+ const tokenRegex = buildTokenRegex(oldLF)
1416
+
1417
+ // Strategy 1: exact literal match.
1418
+ const exactOccurrences = countOccurrences(currentContentLF, oldLF)
1419
+ if (exactOccurrences === expectedReplacements) {
1420
+ currentContentLF = safeLiteralReplace(currentContentLF, oldLF, newLF)
1421
+ } else {
1422
+ // Strategy 2: whitespace-tolerant regex.
1423
+ const wsOccurrences = countRegexMatches(currentContentLF, wsRegex)
1424
+ if (wsOccurrences === expectedReplacements) {
1425
+ currentContentLF = currentContentLF.replace(wsRegex, () => newLF)
1426
+ } else {
1427
+ // Strategy 3: token-based regex.
1428
+ const tokenOccurrences = countRegexMatches(currentContentLF, tokenRegex)
1429
+ if (tokenOccurrences === expectedReplacements) {
1430
+ currentContentLF = currentContentLF.replace(tokenRegex, () => newLF)
1431
+ } else {
1432
+ const anyMatches = exactOccurrences > 0 || wsOccurrences > 0 || tokenOccurrences > 0
1433
+ if (!anyMatches) {
1434
+ return err(
1435
+ `edit_file: no match found in file '${rel}'\n\n<error_details>\nThe provided old_string could not be found using exact, whitespace-tolerant, or token-based matching.\n\nRecovery suggestions:\n1. Use read_file to confirm the file's current contents\n2. Ensure old_string matches exactly (including whitespace/indentation and line endings)\n3. Provide more surrounding context in old_string to make the match unique\n4. If the file has changed since you constructed old_string, re-read and retry\n</error_details>`,
1436
+ )
1437
+ }
1438
+ if (exactOccurrences > 0) {
1439
+ return err(
1440
+ `edit_file: occurrence count mismatch in file '${rel}'\n\n<error_details>\nExpected ${expectedReplacements} occurrence(s) but found ${exactOccurrences} exact match(es).\n\nRecovery suggestions:\n1. Provide a more specific old_string so it matches exactly once — this is almost always the right fix; a short old_string like a bare identifier matches every place that name is USED, not just the one place you want to change\n2. Only set expected_replacements to ${exactOccurrences} if new_string is a genuine like-for-like replacement (e.g. a rename) that is EQUALLY correct at all ${exactOccurrences} locations — an insertion or a large addition is essentially never correct at multiple sites\n3. Use read_file to confirm the exact text and counts\n</error_details>`,
1441
+ )
1442
+ }
1443
+ return err(
1444
+ `edit_file: occurrence count mismatch in file '${rel}'\n\n<error_details>\nExpected ${expectedReplacements} occurrence(s), but matching found ${wsOccurrences} (whitespace-tolerant) and ${tokenOccurrences} (token-based).\n\nRecovery suggestions:\n1. Provide more surrounding context in old_string to make the match unique — this is almost always the right fix\n2. Only adjust expected_replacements to match multiple sites if new_string is EQUALLY correct at every one of them (e.g. a rename) — never for an insertion or addition\n3. Use read_file to confirm the current file contents and refine the match\n</error_details>`,
1445
+ )
1446
+ }
1447
+ }
1448
+ }
1449
+ }
1450
+
1451
+ // Apply the replacement (creating the file when old_string was "").
1452
+ const newContent = isNewFile
1453
+ ? newString
1454
+ : restoreLineEnding(currentContentLF ?? currentContent ?? "", originalEol)
1455
+
1456
+ if (!isNewFile && newContent === currentContent) {
1457
+ return ok(`No changes needed for '${rel}'`)
1458
+ }
1459
+
1460
+ try {
1461
+ await fsp.mkdir(path.dirname(target), { recursive: true })
1462
+ await fsp.writeFile(target, newContent, "utf-8")
1463
+ } catch (error) {
1464
+ return err(`edit_file: failed to write '${rel}': ${errorMessage(error)}`)
1465
+ }
1466
+
1467
+ const replacementInfo = !isNewFile && expectedReplacements > 1 ? ` (${expectedReplacements} replacements)` : ""
1468
+ return ok(`${isNewFile ? `File created: ${rel}` : `File updated: ${rel}`}${replacementInfo}`)
1469
+ })()
1470
+ }
1471
+
1472
+ /**
1473
+ * set_indentation — change ONE line's leading indentation to an exact tab
1474
+ * count, given as a plain integer rather than a literal whitespace string
1475
+ * (issue #141). edit_file's old_string/new_string already tolerates
1476
+ * whitespace-amount differences via buildWhitespaceTolerantRegex, but that
1477
+ * only helps once the MATCH succeeds — live-verified 2026-08-21 that a
1478
+ * local model asked to fix a pure-indentation mismatch, even given the
1479
+ * exact current and desired content verbatim, sometimes submits an
1480
+ * old_string byte-identical to new_string (refused by the "no changes to
1481
+ * apply" check below `oldLF === newLF`) rather than actually varying the
1482
+ * leading whitespace between the two multi-line strings it has to type
1483
+ * out. An integer tab count sidesteps the problem structurally: there is
1484
+ * no pair of near-identical multi-line strings for the model to get
1485
+ * subtly wrong, just one small number.
1486
+ */
1487
+ async function setIndentationHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
1488
+ const filePath = requireString(args, "path")
1489
+ const line = toNonNegativeInt(args.line, 0)
1490
+ const tabs = toNonNegativeInt(args.tabs, -1)
1491
+
1492
+ if (line < 1) {
1493
+ return err(`set_indentation: 'line' must be a positive integer (1-indexed), got ${String(args.line)}`)
1494
+ }
1495
+ if (tabs < 0) {
1496
+ return err(`set_indentation: 'tabs' must be a non-negative integer, got ${String(args.tabs)}`)
1497
+ }
1498
+
1499
+ return (async () => {
1500
+ const target = safeTarget(ctx, filePath)
1501
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target)
1502
+
1503
+ const refusal = protectedWriteRefusal("set_indentation", rel, ctx.permissions)
1504
+ if (refusal !== null) {
1505
+ return refusal
1506
+ }
1507
+
1508
+ let content: string
1509
+ try {
1510
+ content = await fsp.readFile(target, "utf-8")
1511
+ } catch (error) {
1512
+ return err(`set_indentation: failed to read file '${rel}': ${errorMessage(error)}`)
1513
+ }
1514
+
1515
+ const eol = detectLineEnding(content)
1516
+ const lines = normalizeToLF(content).split("\n")
1517
+ if (line > lines.length) {
1518
+ return err(
1519
+ `set_indentation: line ${line} does not exist — '${rel}' has ${lines.length} line(s).\n\nRecovery suggestions:\n1. Use read_file to confirm the real line number\n2. If the file has changed since you last read it, re-read and retry`,
1520
+ )
1521
+ }
1522
+
1523
+ const targetLine = lines[line - 1] ?? ""
1524
+ const currentIndentMatch = /^[\t ]*/.exec(targetLine)
1525
+ const currentIndent = currentIndentMatch ? currentIndentMatch[0] : ""
1526
+ const rest = targetLine.slice(currentIndent.length)
1527
+ const newIndent = "\t".repeat(tabs)
1528
+
1529
+ if (currentIndent === newIndent) {
1530
+ return err(
1531
+ `set_indentation: line ${line} of '${rel}' already has exactly ${tabs} leading tab(s) — no change to make.\n\nRecovery suggestions:\n1. Use read_file to confirm the real current indentation\n2. If a different line needs the fix, check the line number`,
1532
+ )
1533
+ }
1534
+
1535
+ lines[line - 1] = newIndent + rest
1536
+ const newContent = restoreLineEnding(lines.join("\n"), eol)
1537
+
1538
+ try {
1539
+ await fsp.writeFile(target, newContent, "utf-8")
1540
+ } catch (error) {
1541
+ return err(`set_indentation: failed to write '${rel}': ${errorMessage(error)}`)
1542
+ }
1543
+
1544
+ return ok(
1545
+ `File updated: ${rel} (line ${line} indentation changed from ${currentIndent.length} to ${tabs} tab${tabs === 1 ? "" : "s"})`,
1546
+ )
1547
+ })()
1548
+ }
1549
+
1550
+ /**
1551
+ * execute_command — spawn via child_process, capture stdout+stderr.
1552
+ *
1553
+ * Timeout semantics follow the vendored tool's own contract (see
1554
+ * src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts):
1555
+ * when the `timeout` elapses the command KEEPS RUNNING in the background — it
1556
+ * is NOT killed — and the model receives the output captured so far so it can
1557
+ * start a dev server / long migration and move on. The result is a normal
1558
+ * (non-error) tool result: a timeout here is expected behavior the model asked
1559
+ * for via the `timeout` arg, so it must not count toward the loop's
1560
+ * consecutive-mistake bounded-failure limit.
1561
+ */
1562
+ function executeCommandHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
1563
+ const command = requireString(args, "command")
1564
+ const timeoutS = toNonNegativeInt(args.timeout, DEFAULT_COMMAND_TIMEOUT_S) || DEFAULT_COMMAND_TIMEOUT_S
1565
+
1566
+ return (async () => {
1567
+ let cwd = ctx.workspaceRoot
1568
+ if (args.cwd != null && args.cwd !== "") {
1569
+ const cwdArg = requireString(args, "cwd")
1570
+ cwd = safeTarget(ctx, cwdArg)
1571
+ }
1572
+
1573
+ // Permissions gate (command allow/deny + dangerous substitution +
1574
+ // central-store protection + redirect-escape): refuse BEFORE spawning
1575
+ // so nothing is ever executed. Compound commands are checked per
1576
+ // sub-command (parseCommand splits on &&/||/;/|/&) — a denied or
1577
+ // unallow-listed sub-command blocks the whole chain. This applies to
1578
+ // every executor built on this handler, including the read-only
1579
+ // reviewer/QA executors (they share it), which also run unattended and
1580
+ // could otherwise run arbitrary commands. A refusal is a real tool
1581
+ // error (isError: true) and therefore counts toward the loop's
1582
+ // consecutive-mistake bound, same as any tool error. The command's
1583
+ // resolved `cwd` anchors relative targets for the central-store and
1584
+ // redirect-escape checks (see src/permissions/store-protection.ts and
1585
+ // src/permissions/commands.ts). The redirect-escape scan runs against
1586
+ // the RAW command before checkCommand parses it (a redirect embedded
1587
+ // in an unparseable fragment — e.g. a heredoc body's `> /tmp` line —
1588
+ // would otherwise be swallowed by the malformed-command path).
1589
+ const redirectRoot = cwd
1590
+ const redirectEscape = checkRedirectEscape(command, redirectRoot)
1591
+ if (redirectEscape !== null) {
1592
+ return refusalMessage(command, { kind: "redirect_escape", subCommand: command, redirect: redirectEscape })
1593
+ }
1594
+ const refusal = checkCommand(command, ctx.permissions.allowedCommands, ctx.permissions.deniedCommands, {
1595
+ workspaceRoot: cwd,
1596
+ })
1597
+ if (refusal !== null) {
1598
+ return refusalMessage(command, refusal)
1599
+ }
1600
+
1601
+ return new Promise<ToolResult>((resolve) => {
1602
+ let stdout = ""
1603
+ let stderr = ""
1604
+ let settled = false
1605
+ let timedOut = false
1606
+
1607
+ const finish = (result: ToolResult) => {
1608
+ if (!settled) {
1609
+ settled = true
1610
+ resolve(result)
1611
+ }
1612
+ }
1613
+
1614
+ // 2026-08-27: verified live, repeatedly, across several otherwise-
1615
+ // healthy long-running sessions (no memory pressure per
1616
+ // /proc/pressure/memory, no zombie/fd accumulation found) — spawn()
1617
+ // intermittently throws ENOENT for the shell itself ("spawn
1618
+ // /bin/bash ENOENT") even though /bin/bash demonstrably exists and
1619
+ // BASH_PATH resolved it correctly at module load. Root cause not
1620
+ // pinned down (a transient Node/libuv spawn hiccup is the leading
1621
+ // theory, not a real missing binary), but the failure mode is
1622
+ // unambiguous: a completely benign command (`git log`, `pwd`) fails
1623
+ // this way, derails the model with a spurious mistake, and burns
1624
+ // through the session's mistake budget on pure infrastructure
1625
+ // noise. One transparent retry (fresh spawn, same command/options)
1626
+ // before surfacing anything to the model treats this as the
1627
+ // transient it appears to be instead of a model-facing error.
1628
+ const isBashSpawnEnoent = (error: unknown): boolean =>
1629
+ error instanceof Error &&
1630
+ "code" in error &&
1631
+ (error as NodeJS.ErrnoException).code === "ENOENT" &&
1632
+ ("path" in error ? String((error as { path?: unknown }).path ?? "") : "").includes("bash")
1633
+
1634
+ // 2026-08-27: with the double-fire bug above fixed (each attempt now
1635
+ // genuinely settles once), the same daemon-review sessions still hit
1636
+ // this on some runs even across the full retry budget — the real
1637
+ // transient window can outlast a ~1.5s total retry span. Widened to
1638
+ // 8 attempts with a longer per-step backoff (500ms * attempt, so the
1639
+ // full span is several seconds) rather than assume 4 attempts was
1640
+ // already enough patience.
1641
+ const MAX_SPAWN_ATTEMPTS = 8
1642
+ const SPAWN_RETRY_DELAY_MS = 500
1643
+ const spawnAttempt = (attempt: number) => {
1644
+ // 2026-08-27: verified live — a failed spawn can fire BOTH the
1645
+ // `error` event AND the `close` event for the SAME underlying
1646
+ // failure (a known Node/libuv behavior: a child that never truly
1647
+ // started still gets its close lifecycle completed). Without a
1648
+ // guard, each independently retried, so one real failure produced
1649
+ // TWO parallel retry chains, each of which could again double —
1650
+ // confirmed live via temporary diagnostic logging: attempt counts
1651
+ // literally doubled at each level (2, 4, 8 duplicate retries for
1652
+ // the same command). That storm of concurrent spawns was plausibly
1653
+ // making the underlying transient WORSE, not better. This attempt's
1654
+ // own outcome (retry-or-finish) must be decided exactly once.
1655
+ let attemptSettled = false
1656
+ let child
1657
+ try {
1658
+ // `detached: true` puts the shell in its own process group
1659
+ // (pgid = child.pid). That serves two purposes: (a) a timed-out
1660
+ // command can keep running fully independent of the harness's
1661
+ // own process group, and (b) session teardown can kill the whole
1662
+ // group (`process.kill(-pid)`) so grandchildren are reaped too.
1663
+ // The process-group kill is the only way to clean up the full
1664
+ // tree of a `shell: true` spawn — killing the shell alone would
1665
+ // orphan whatever it had launched.
1666
+ child = spawn(command, {
1667
+ cwd,
1668
+ // `shell: true` alone defaults to `/bin/sh` (dash on Debian/
1669
+ // Ubuntu, the common host here), which has no bash-only
1670
+ // features — `${PIPESTATUS[0]}`, `[[ ]]`, arrays. A model that
1671
+ // reaches for one of these (common; every mode's rules files
1672
+ // are silent on which shell dialect execute_command actually
1673
+ // runs) gets a shell PARSE error on the whole command line —
1674
+ // including whatever real command preceded it (e.g.
1675
+ // `npm test 2>&1 | tail -60; echo "EXIT=${PIPESTATUS[0]}"`)
1676
+ // fails with exit 2 even though `npm test` itself may have
1677
+ // passed cleanly. Confirmed live 2026-08-05: this produced a
1678
+ // genuine FALSE "QA_VERDICT: FAIL" on issue #17's round — a
1679
+ // manual re-run of the identical code passed cleanly. Every
1680
+ // `scripts/*.sh` in this repo already assumes bash; making
1681
+ // execute_command match removes an entire class of spurious
1682
+ // tool/verification failures instead of just working around
1683
+ // each occurrence as it's spotted.
1684
+ shell: BASH_PATH ?? true,
1685
+ detached: true,
1686
+ // Matches the vendored Execa-based terminal's own spawn options
1687
+ // (zoo-code/src/integrations/terminal/ExecaTerminalProcess.ts):
1688
+ // ignore stdin so a command that reads from it gets an immediate
1689
+ // EOF instead of hanging on an open, never-written pipe (there is
1690
+ // no interactive user here to type anything), and force a UTF-8
1691
+ // locale so tools sensitive to it (Ruby, CocoaPods, etc. per the
1692
+ // vendored comment) behave consistently regardless of the host's
1693
+ // own locale configuration.
1694
+ stdio: ["ignore", "pipe", "pipe"],
1695
+ env: { ...process.env, LANG: "en_US.UTF-8", LC_ALL: "en_US.UTF-8" },
1696
+ })
1697
+ } catch (error) {
1698
+ if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
1699
+ // A live standalone repro of one of these exact failures spawned
1700
+ // cleanly on the first try outside the long-running harness
1701
+ // process — this is not an inherently broken command, so a
1702
+ // single immediate retry was landing in the same narrow window
1703
+ // as the original failure. Verified live 2026-08-27: failures
1704
+ // can come in bursts of several consecutive calls, not just an
1705
+ // isolated blip (plausibly a GC pause in this specific
1706
+ // long-running process interacting with posix_spawn) — one
1707
+ // retry wasn't always enough. Escalating delay across up to
1708
+ // MAX_SPAWN_ATTEMPTS gives a longer transient window room to
1709
+ // clear before this surfaces to the model as a real failure.
1710
+ setTimeout(() => spawnAttempt(attempt + 1), SPAWN_RETRY_DELAY_MS * attempt)
1711
+ return
1712
+ }
1713
+ finish(err(`execute_command: failed to spawn '${command}': ${errorMessage(error)}`))
1714
+ return
1715
+ }
1716
+
1717
+ const timer = setTimeout(() => {
1718
+ // Do NOT kill the child — the vendored contract says a timed-out
1719
+ // command keeps running in the background so the model can start
1720
+ // dev servers / long migrations and get control back.
1721
+ timedOut = true
1722
+ backgroundCommands.add(child)
1723
+ // Detach from the harness's event loop: the child and its pipes
1724
+ // must not keep the process alive once a session is done. The
1725
+ // child stays tracked in backgroundCommands so session teardown
1726
+ // (ToolExecutor.dispose) can hard-kill it rather than orphan it.
1727
+ child.unref()
1728
+ unrefStream(child.stdout)
1729
+ unrefStream(child.stderr)
1730
+ const combined = [stdout, stderr].filter(Boolean).join("\n")
1731
+ finish(
1732
+ ok(
1733
+ `[execute_command] timed out after ${timeoutS}s — process is still running in the background, output captured so far:\n${combined}`,
1734
+ ),
1735
+ )
1736
+ }, timeoutS * 1000)
1737
+
1738
+ child.stdout?.on("data", (chunk) => {
1739
+ // After a timeout the model already got its partial output; keep
1740
+ // draining the pipe (so the child never blocks on a full buffer)
1741
+ // but stop accumulating output we can no longer deliver. Once a
1742
+ // stream hits its cap, keep draining but discard — the final
1743
+ // truncation/summarization at close stays the single cut point.
1744
+ if (!timedOut && stdout.length < MAX_COMMAND_STREAM_CHARS) {
1745
+ stdout += chunk.toString()
1746
+ }
1747
+ })
1748
+ child.stderr?.on("data", (chunk) => {
1749
+ if (!timedOut && stderr.length < MAX_COMMAND_STREAM_CHARS) {
1750
+ stderr += chunk.toString()
1751
+ }
1752
+ })
1753
+ child.on("error", (error) => {
1754
+ if (attemptSettled) {
1755
+ return
1756
+ }
1757
+ attemptSettled = true
1758
+ clearTimeout(timer)
1759
+ backgroundCommands.delete(child)
1760
+ if (attempt < MAX_SPAWN_ATTEMPTS && isBashSpawnEnoent(error)) {
1761
+ stdout = ""
1762
+ stderr = ""
1763
+ setTimeout(() => spawnAttempt(attempt + 1), SPAWN_RETRY_DELAY_MS * attempt)
1764
+ return
1765
+ }
1766
+ finish(err(`execute_command: spawn error for '${command}': ${errorMessage(error)}`))
1767
+ })
1768
+ child.on("close", (code, signal) => {
1769
+ if (attemptSettled) {
1770
+ return
1771
+ }
1772
+ attemptSettled = true
1773
+ clearTimeout(timer)
1774
+ backgroundCommands.delete(child)
1775
+ const combined = [stdout, stderr].filter(Boolean).join("\n")
1776
+ // Log debug information
1777
+ if (process.env.HEADLESSCODE_DEBUG) {
1778
+ console.error(`[execute_command debug] Command: ${command}, Code: ${code}, Signal: ${signal}, stdout: ${stdout.slice(0, 200)}, stderr: ${stderr.slice(0, 200)}`)
1779
+ }
1780
+ // 2026-08-27: the SAME bash-spawn-ENOENT failure this file already
1781
+ // retries on (see isBashSpawnEnoent / the `error` handler above)
1782
+ // was found live to ALSO surface via a completely different path --
1783
+ // no `error` event at all, just `close` firing with `code: -2`
1784
+ // (Node's posix_spawn fast path reporting a negated errno directly:
1785
+ // -2 is -ENOENT). A retry that only listens for the `error` event
1786
+ // silently misses this shape entirely. `code` is negative here in
1787
+ // no other real scenario (a genuine command exit code is always
1788
+ // 0-255), so treat any negative code on the first attempt the same
1789
+ // way: retry once before surfacing anything to the model.
1790
+ if (attempt < MAX_SPAWN_ATTEMPTS && code !== null && code < 0) {
1791
+ stdout = ""
1792
+ stderr = ""
1793
+ setTimeout(() => spawnAttempt(attempt + 1), SPAWN_RETRY_DELAY_MS * attempt)
1794
+ return
1795
+ }
1796
+ if (signal === "SIGKILL") {
1797
+ // Defensive path only: a timeout never sends SIGKILL anymore,
1798
+ // so this is reachable solely when something external killed
1799
+ // the process (the model's own `pkill` cleanup command, or
1800
+ // ToolExecutor.dispose during session teardown).
1801
+ finish(err(`execute_command: command '${command}' was killed by signal ${signal}.\n${combined}`))
1802
+ return
1803
+ }
1804
+ if (code !== 0) {
1805
+ // Command failed: the stderr/stdout tail is exactly what the
1806
+ // model needs. Summarize it (opt-in) exactly like the success
1807
+ // path — a failed verbose test run is THE canonical case.
1808
+ // .catch (issue #82): summarizeCommandOutput never rejects today,
1809
+ // but this is a fire-and-forget .then() with no caller to
1810
+ // propagate a rejection to — an unhandled rejection would crash
1811
+ // the whole process on Node 15+ if that invariant ever breaks.
1812
+ // Fall back to the raw (unsummarized) output rather than lose
1813
+ // the result.
1814
+ void summarizeCommandOutput(combined)
1815
+ .then((content) =>
1816
+ finish(err(`execute_command: command '${command}' exited with code ${code}.\n${content}`)),
1817
+ )
1818
+ .catch(() =>
1819
+ finish(err(`execute_command: command '${command}' exited with code ${code}.\n${combined}`)),
1820
+ )
1821
+ return
1822
+ }
1823
+ void summarizeCommandOutput(combined === "" ? `(command completed with no output)` : combined)
1824
+ .then((content) => finish(ok(content)))
1825
+ .catch(() => finish(ok(combined === "" ? `(command completed with no output)` : combined)))
1826
+ })
1827
+ }
1828
+
1829
+ spawnAttempt(1)
1830
+ })
1831
+ })()
1832
+ }
1833
+
1834
+ /** One list_files repeat-call guard entry — see listFilesHandler's doc comment. */
1835
+ type ListFilesCallEntry = {
1836
+ /** The condensation generation this key was last (really) listed at. */
1837
+ generation: number
1838
+ /**
1839
+ * Whether the LAST identical call already got the short cache-hit
1840
+ * message instead of a real listing. When true, the next identical
1841
+ * call gets a real listing again — mirrors read_file's toldUnchanged
1842
+ * (readFileHandler) exactly, and for the same reason: a one-shot notice
1843
+ * never loops forever.
1844
+ */
1845
+ toldUnchanged: boolean
1846
+ }
1847
+
1848
+ /**
1849
+ * list_files — top-level or recursive listing, dirs first, truncated.
1850
+ *
1851
+ * Repeat-call guard: a local model was observed calling list_files with the
1852
+ * SAME (path, recursive) twice in one turn pair with no condensation in
1853
+ * between (verified live 2026-08-20 — see
1854
+ * plans/local-dual-model-code-agent-PROMPT-2026-08-21.md's trial 4), which
1855
+ * cost ~7.5K prompt tokens for a result already sitting in context. When
1856
+ * `calls`/`generation` are supplied (see ToolExecutor.registerListFiles), an
1857
+ * identical call within the same condensation generation gets a short
1858
+ * cache-hit notice INSTEAD of a refusal — not an error, so it never counts
1859
+ * against the consecutive-mistake budget. This was originally a hard
1860
+ * refusal (matching guardLargeOverwrites' shape); verified live 2026-08-20
1861
+ * that this was actively harmful — a model that repeats an identical call
1862
+ * verbatim after an error (the SAME cross-tool tic already documented for
1863
+ * apply_diff/ask_followup_question, see the plan doc above) burned an
1864
+ * entire 8-mistake budget refusing the identical list_files call 8 times in
1865
+ * a row and never reached the real task. A repeat AFTER a condensation is
1866
+ * always treated as fresh: the earlier result may have been the part that
1867
+ * got compressed away.
1868
+ */
1869
+ function listFilesHandler(
1870
+ args: Record<string, unknown>,
1871
+ ctx: ToolContext,
1872
+ calls?: Map<string, ListFilesCallEntry>,
1873
+ generation?: number,
1874
+ ): Promise<ToolResult> {
1875
+ const dirPath = args.path == null || args.path === "" ? "." : requireString(args, "path")
1876
+ const recursive = args.recursive === true
1877
+
1878
+ return (async () => {
1879
+ const target = safeTarget(ctx, dirPath)
1880
+ const rel = path.relative(ctx.workspaceRoot, target).toPosix() || path.basename(target) || "."
1881
+ const key = `${target} ${recursive}`
1882
+ const entry = calls?.get(key)
1883
+
1884
+ if (calls !== undefined && generation !== undefined && entry !== undefined && entry.generation === generation) {
1885
+ if (!entry.toldUnchanged) {
1886
+ entry.toldUnchanged = true
1887
+ return ok(
1888
+ `[cache] '${rel}' (recursive=${recursive}) was already listed earlier this session and nothing has been ` +
1889
+ `condensed since — reuse the earlier result above instead of re-listing. Re-listing again will re-run ` +
1890
+ `the real listing.`,
1891
+ )
1892
+ }
1893
+ // Fall through to a real listing: a second identical call in a row
1894
+ // means the cache-hit notice alone didn't redirect the model, and
1895
+ // refusing again would just repeat the same unproductive exchange.
1896
+ }
1897
+
1898
+ let collected: { entries: Array<{ rel: string; isDir: boolean }>; truncated: boolean }
1899
+ try {
1900
+ collected = await collectEntries(target, ctx.workspaceRoot, recursive)
1901
+ } catch (error) {
1902
+ return err(`list_files: cannot list '${rel}': ${errorMessage(error)}`)
1903
+ }
1904
+
1905
+ // Only recorded on SUCCESS: a failed listing (e.g. bad path) should
1906
+ // remain retryable with the real error, never masked by the cache-hit
1907
+ // notice above.
1908
+ if (calls !== undefined && generation !== undefined) {
1909
+ calls.set(key, { generation, toldUnchanged: false })
1910
+ }
1911
+
1912
+ // Sort dirs first, then alphabetically (over the collected subset).
1913
+ collected.entries.sort((a, b) => {
1914
+ if (a.isDir !== b.isDir) {
1915
+ return a.isDir ? -1 : 1
1916
+ }
1917
+ return a.rel.localeCompare(b.rel)
1918
+ })
1919
+
1920
+ const lines = collected.entries.slice(0, MAX_LIST_FILES).map((e) => (e.isDir ? `${e.rel}/` : e.rel))
1921
+ // Once the walk was capped the true total is unknown, so the trailer
1922
+ // omits it — the trailer itself is still the signal that more exists.
1923
+ const truncatedNote = collected.truncated
1924
+ ? `\n(File list truncated: ${MAX_LIST_FILES} entries shown. Use list_files on specific subdirectories to see more.)`
1925
+ : ""
1926
+
1927
+ return ok(lines.length > 0 ? lines.join("\n") + truncatedNote : "(empty directory)")
1928
+ })()
1929
+ }
1930
+
1931
+ async function collectEntries(
1932
+ dir: string,
1933
+ root: string,
1934
+ recursive: boolean,
1935
+ depth = 0,
1936
+ budget = MAX_LIST_FILES,
1937
+ ): Promise<{ entries: Array<{ rel: string; isDir: boolean }>; truncated: boolean }> {
1938
+ if (depth > 32) {
1939
+ return { entries: [], truncated: false }
1940
+ }
1941
+ const dirents = await fsp.readdir(dir, { withFileTypes: true })
1942
+ const out: Array<{ rel: string; isDir: boolean }> = []
1943
+ let truncated = false
1944
+ for (const ent of dirents) {
1945
+ // Budget exhausted: stop pushing and stop descending — the caller only
1946
+ // keeps the first MAX_LIST_FILES anyway, so a huge tree is never fully
1947
+ // walked, collected, or sorted.
1948
+ if (out.length >= budget) {
1949
+ truncated = true
1950
+ break
1951
+ }
1952
+ const abs = path.join(dir, ent.name)
1953
+ const rel = path.relative(root, abs).toPosix() || ent.name
1954
+ if (ent.isDirectory()) {
1955
+ out.push({ rel, isDir: true })
1956
+ if (recursive) {
1957
+ const sub = await collectEntries(abs, root, recursive, depth + 1, budget - out.length)
1958
+ out.push(...sub.entries)
1959
+ truncated = truncated || sub.truncated
1960
+ }
1961
+ } else {
1962
+ out.push({ rel, isDir: false })
1963
+ }
1964
+ }
1965
+ return { entries: out, truncated }
1966
+ }
1967
+
1968
+ /** attempt_completion — the loop intercepts this; registering it keeps the executor total. */
1969
+ function attemptCompletionHandler(args: Record<string, unknown>): ToolResult {
1970
+ const result = typeof args.result === "string" ? args.result : JSON.stringify(args)
1971
+ return ok(result)
1972
+ }
1973
+
1974
+ /** Marker basenames, relative to the workspace root (mirrors the .harness.* idiom). */
1975
+ export const NEEDS_DECISION_FILENAME = ".harness.needs-decision"
1976
+ export const DECISION_ANSWER_FILENAME = ".harness.decision-answer"
1977
+
1978
+ /** Default ask_followup_question escalation timeout: 30 minutes. */
1979
+ export const DEFAULT_DECISION_TIMEOUT_MS = 1_800_000
1980
+ /** Default poll interval while waiting for an answer, matching watch.ts's DEFAULT_POLL_INTERVAL_MS. */
1981
+ export const DEFAULT_DECISION_POLL_INTERVAL_MS = 5_000
1982
+
1983
+ /** Today's non-interactive fallback text — reused verbatim on timeout (see file header). */
1984
+ function autonomousDecisionError(question: string): ToolResult {
1985
+ return {
1986
+ content: `[Non-interactive] This headless harness cannot display questions or collect answers. The model must decide autonomously. (Question asked: ${question})`,
1987
+ isError: true,
1988
+ }
1989
+ }
1990
+
1991
+ /** Extract suggestion text from the `follow_up` arg (see the vendored native-tool schema). */
1992
+ function extractSuggestions(followUp: unknown): string[] | undefined {
1993
+ if (!Array.isArray(followUp)) {
1994
+ return undefined
1995
+ }
1996
+ const texts = followUp
1997
+ .map((item) => (item !== null && typeof item === "object" ? (item as Record<string, unknown>).text : undefined))
1998
+ .filter((t): t is string => typeof t === "string" && t.length > 0)
1999
+ return texts.length > 0 ? texts : undefined
2000
+ }
2001
+
2002
+ async function safeUnlink(p: string): Promise<void> {
2003
+ try {
2004
+ await fsp.unlink(p)
2005
+ } catch {
2006
+ // Already gone / never existed — fine either way.
2007
+ }
2008
+ }
2009
+
2010
+ /** The terminal outcome of one decision-escalation wait (see escalateDecision). */
2011
+ export type DecisionEscalationResult =
2012
+ | { status: "answered"; answer: string }
2013
+ | { status: "timedOut" }
2014
+ | { status: "writeFailed"; error: string }
2015
+
2016
+ /**
2017
+ * Shared decision-escalation primitive (ask_followup_question AND switch_mode
2018
+ * — see plans/switch-mode-headless.md; do NOT duplicate this for future
2019
+ * blocking tools).
2020
+ *
2021
+ * Writes `<workspaceRoot>/.harness.needs-decision` (JSON: question,
2022
+ * suggestions?, askedAt), then polls for `<workspaceRoot>/.harness.decision-
2023
+ * answer` every `decisionPollIntervalMs` (default 5s) up to
2024
+ * `decisionTimeoutMs` (default 30min). The session's budget-duration clock is
2025
+ * paused for the duration of the wait (see pauseBudgetClock/resumeBudgetClock
2026
+ * on ToolContext). Both marker files are ALWAYS cleaned up on every terminal
2027
+ * outcome, and the decision_blocked / decision_answered event hooks fire at
2028
+ * the same lifecycle points as before.
2029
+ *
2030
+ * The CALLER decides what each outcome means for the model:
2031
+ * - answered: a human/orchestrator wrote the answer file; the raw answer
2032
+ * text (trimmed) is returned for the caller to interpret.
2033
+ * - timedOut: no answer arrived within the timeout; the marker is already
2034
+ * cleaned up, so a caller that refuses on timeout leaves no debris.
2035
+ * - writeFailed: the needs-decision marker itself couldn't be written (e.g.
2036
+ * a read-only workspace) — the caller should fail immediately rather than
2037
+ * block on an answer no one can ever provide.
2038
+ */
2039
+ export async function escalateDecision(
2040
+ ctx: ToolContext,
2041
+ question: string,
2042
+ suggestions?: string[],
2043
+ ): Promise<DecisionEscalationResult> {
2044
+ const needsDecisionPath = path.join(ctx.workspaceRoot, NEEDS_DECISION_FILENAME)
2045
+ const answerPath = path.join(ctx.workspaceRoot, DECISION_ANSWER_FILENAME)
2046
+ const timeoutMs = ctx.decisionTimeoutMs ?? DEFAULT_DECISION_TIMEOUT_MS
2047
+ const pollIntervalMs = ctx.decisionPollIntervalMs ?? DEFAULT_DECISION_POLL_INTERVAL_MS
2048
+ const askedAt = new Date().toISOString()
2049
+
2050
+ try {
2051
+ await fsp.writeFile(
2052
+ needsDecisionPath,
2053
+ JSON.stringify({ question, ...(suggestions ? { suggestions } : {}), askedAt }, null, 2) + "\n",
2054
+ "utf-8",
2055
+ )
2056
+ } catch (error) {
2057
+ // Can't even signal escalation (e.g. read-only workspace) — report the
2058
+ // failure and let the caller decide, rather than blocking on a wait no
2059
+ // one can ever answer.
2060
+ return { status: "writeFailed", error: errorMessage(error) }
2061
+ }
2062
+ // Live worker monitoring: mirror the marker write on the session's event
2063
+ // feed (see src/engine/events.ts). Non-fatal — a feed failure must never
2064
+ // affect the tool result.
2065
+ ctx.onDecisionEvent?.("decision_blocked", { question, ...(suggestions ? { suggestions } : {}) })
2066
+
2067
+ ctx.pauseBudgetClock?.()
2068
+ let answered = false
2069
+ try {
2070
+ const deadline = Date.now() + timeoutMs
2071
+ while (Date.now() < deadline) {
2072
+ let answer: string | undefined
2073
+ try {
2074
+ answer = await fsp.readFile(answerPath, "utf-8")
2075
+ } catch {
2076
+ answer = undefined
2077
+ }
2078
+ if (answer !== undefined) {
2079
+ await safeUnlink(needsDecisionPath)
2080
+ await safeUnlink(answerPath)
2081
+ ctx.onDecisionEvent?.("decision_answered", { answer: answer.trim() })
2082
+ answered = true
2083
+ return { status: "answered", answer: answer.trim() }
2084
+ }
2085
+ const remaining = deadline - Date.now()
2086
+ if (remaining <= 0) {
2087
+ break
2088
+ }
2089
+ await sleep(Math.min(pollIntervalMs, remaining))
2090
+ }
2091
+ } finally {
2092
+ ctx.resumeBudgetClock?.()
2093
+ }
2094
+ if (!answered) {
2095
+ ctx.onDecisionEvent?.("decision_answered", { timedOut: true })
2096
+ }
2097
+ // Timed out: clean up BOTH markers (BUG-3). The answer file is usually
2098
+ // absent here, but a write can land just after the poll loop's last read
2099
+ // and before this point (a narrow but real race — the poller's last
2100
+ // `readFile` above can lose to a concurrent writer by a few ms). If that
2101
+ // happens and only needsDecisionPath were unlinked, the stale answerPath
2102
+ // would sit on disk and get read as the answer to a LATER, unrelated
2103
+ // escalateDecision call (the next ask_followup_question/switch_mode, or a
2104
+ // later session reusing the same worktree) — poisoning it with a stale
2105
+ // answer for the wrong question. Unlinking both here, unconditionally,
2106
+ // closes that gap.
2107
+ await safeUnlink(needsDecisionPath)
2108
+ await safeUnlink(answerPath)
2109
+ return { status: "timedOut" }
2110
+ }
2111
+
2112
+ /**
2113
+ * ask_followup_question — decision escalation (workstream 2). Thin wrapper
2114
+ * over the shared escalateDecision primitive: the semantics are UNCHANGED
2115
+ * from before the refactor.
2116
+ *
2117
+ * - Answer arrives in time: the answer text is returned as a NORMAL
2118
+ * (non-error) tool result — the model continues with real input, and this
2119
+ * does NOT count toward the loop's consecutive-mistake bounded-failure
2120
+ * limit.
2121
+ * - Timeout elapses (or the marker couldn't even be written): the handler
2122
+ * falls back to EXACTLY today's behavior — the "must decide autonomously"
2123
+ * tool error, which DOES count as a mistake.
2124
+ */
2125
+ async function askFollowupQuestionHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
2126
+ const question = typeof args.question === "string" ? args.question : "(no question)"
2127
+ const suggestions = extractSuggestions(args.follow_up)
2128
+ const timeoutMs = ctx.decisionTimeoutMs ?? DEFAULT_DECISION_TIMEOUT_MS
2129
+
2130
+ const outcome = await escalateDecision(ctx, question, suggestions)
2131
+ if (outcome.status === "writeFailed") {
2132
+ process.stderr.write(
2133
+ `[headlesscode] ask_followup_question: failed to write ${NEEDS_DECISION_FILENAME}, falling back to autonomous decision: ${outcome.error}\n`,
2134
+ )
2135
+ return autonomousDecisionError(question)
2136
+ }
2137
+ if (outcome.status === "answered") {
2138
+ return ok(`[Human/orchestrator answer received] ${outcome.answer}`)
2139
+ }
2140
+
2141
+ // Timed out: fall back to exactly today's behavior.
2142
+ process.stderr.write(
2143
+ `[headlesscode] ask_followup_question: timed out after ${timeoutMs}ms waiting for ${DECISION_ANSWER_FILENAME}, falling back to autonomous decision.\n`,
2144
+ )
2145
+ return autonomousDecisionError(question)
2146
+ }
2147
+
2148
+ // ─── update_todo_list (structured planning aid) ─────────────────────────────
2149
+
2150
+ /** Status of one checklist line — matches the vendored tool's vocabulary. */
2151
+ export type TodoStatus = "pending" | "in_progress" | "completed"
2152
+
2153
+ /** One parsed checklist item. */
2154
+ export interface TodoItem {
2155
+ content: string
2156
+ status: TodoStatus
2157
+ }
2158
+
2159
+ /**
2160
+ * Full session todo state, as surfaced by `onTodoEvent` / `getTodoList()`.
2161
+ * `todos` is the normalized full checklist (every line, exactly as stored).
2162
+ */
2163
+ export interface TodoListSnapshot {
2164
+ todos: string
2165
+ done: number
2166
+ inProgress: number
2167
+ pending: number
2168
+ }
2169
+
2170
+ /**
2171
+ * Session-scoped todo list state. update_todo_list always REPLACES the whole
2172
+ * list (per the vendored tool contract: "Always provide the full list; the
2173
+ * system will overwrite the previous one"), so state is just the latest
2174
+ * parsed checklist + derived counts. In-memory conversational state only —
2175
+ * never written to a workspace file.
2176
+ */
2177
+ class TodoListState {
2178
+ todos = ""
2179
+ done = 0
2180
+ inProgress = 0
2181
+ pending = 0
2182
+
2183
+ snapshot(): TodoListSnapshot | undefined {
2184
+ if (this.todos === "") {
2185
+ return undefined
2186
+ }
2187
+ return { todos: this.todos, done: this.done, inProgress: this.inProgress, pending: this.pending }
2188
+ }
2189
+
2190
+ replace(todos: string, done: number, inProgress: number, pending: number): void {
2191
+ this.todos = todos
2192
+ this.done = done
2193
+ this.inProgress = inProgress
2194
+ this.pending = pending
2195
+ }
2196
+ }
2197
+
2198
+ /**
2199
+ * Parse a markdown checklist into items + the normalized list. Tolerant of
2200
+ * `-`/`*`/`+` or no list marker, and of `[ ]` / `[x]` / `[X]` / `[-]`.
2201
+ * Lines that don't look like checklist items are preserved in the normalized
2202
+ * output (so an update never silently drops content) but excluded from the
2203
+ * counts, mirroring the vendored tool's single-level checklist format.
2204
+ */
2205
+ export function parseTodoList(todos: string): { items: TodoItem[]; normalized: string } {
2206
+ const items: TodoItem[] = []
2207
+ const lines: string[] = []
2208
+ for (const raw of todos.split(/\r?\n/)) {
2209
+ const line = raw.trimEnd()
2210
+ lines.push(line)
2211
+ const m = /^\s*(?:[-*+]\s+)?\[(.)\]\s*(.*)$/.exec(line)
2212
+ if (!m) {
2213
+ continue
2214
+ }
2215
+ const marker = m[1]
2216
+ let status: TodoStatus = "pending"
2217
+ if (marker === "x" || marker === "X") {
2218
+ status = "completed"
2219
+ } else if (marker === "-") {
2220
+ status = "in_progress"
2221
+ }
2222
+ const content = m[2].trim()
2223
+ if (content !== "") {
2224
+ items.push({ content, status })
2225
+ }
2226
+ }
2227
+ return { items, normalized: lines.join("\n") }
2228
+ }
2229
+
2230
+ /**
2231
+ * The update_todo_list handler (vendored schema: `{todos: string}`, strict,
2232
+ * required). Stores the full checklist as session state and echoes the
2233
+ * normalized list + counts back, so the model sees exactly what was stored.
2234
+ * Fires `onTodoEvent` on every call so the session can surface a
2235
+ * `todo_updated` feed event for observability.
2236
+ */
2237
+ function updateTodoListHandler(args: Record<string, unknown>, ctx: ToolContext, state: TodoListState): ToolResult {
2238
+ const todos = requireString(args, "todos")
2239
+ const { items, normalized } = parseTodoList(todos)
2240
+ const done = items.filter((i) => i.status === "completed").length
2241
+ const inProgress = items.filter((i) => i.status === "in_progress").length
2242
+ const pending = items.filter((i) => i.status === "pending").length
2243
+
2244
+ state.replace(normalized, done, inProgress, pending)
2245
+
2246
+ ctx.onTodoEvent?.({ todos: normalized, done, inProgress, pending })
2247
+
2248
+ return ok(`TODO list updated (${done} completed, ${inProgress} in progress, ${pending} pending):\n${normalized}`)
2249
+ }
2250
+
2251
+ /** Stub for vendored tools that the headless harness does not implement. */
2252
+ function stubHandler(name: string, implemented: readonly string[] = ["read_file", "write_to_file", "apply_diff", "search_replace", "edit_file", "execute_command", "list_files", "codebase_search"]): ToolHandler {
2253
+ const available = [...implemented].sort().join(", ")
2254
+ return () =>
2255
+ err(
2256
+ `Tool '${name}' is not implemented in the headless harness yet. Implemented tools: ${available}. Adapt and retry with one of those.`,
2257
+ )
2258
+ }
2259
+
2260
+ /**
2261
+ * codebase_search — semantic search over the persisted codebase index.
2262
+ *
2263
+ * Embeds the query (one call), loads the central project store's
2264
+ * `codesearch/index.jsonl` (see src/project-store.ts — every worktree of a
2265
+ * repo shares the same index), brute-force cosine against every stored chunk,
2266
+ * and returns the top-K as `file:startLine-endLine` + a snippet — matching the
2267
+ * vendored CodebaseSearchTool's output shape (src/vendor/zoo-code/src/core/
2268
+ * tools/CodebaseSearchTool.ts).
2269
+ *
2270
+ * If no index exists yet, returns a clear ACTIONABLE error telling the model
2271
+ * to ask a human to run `headlesscode index` — deliberately NOT a silent
2272
+ * empty result ("no matches" would look like a legitimately empty search).
2273
+ * The index is built by a separate, explicit CLI step and is never
2274
+ * auto-triggered mid-session (it costs real money and takes real time).
2275
+ */
2276
+ async function codebaseSearchHandler(args: Record<string, unknown>, ctx: ToolContext): Promise<ToolResult> {
2277
+ const query = requireString(args, "query")
2278
+ const pathPrefix = typeof args.path === "string" && args.path.trim() !== "" ? args.path : undefined
2279
+
2280
+ return (async () => {
2281
+ // No index → actionable error, not a silent empty result.
2282
+ const indexFile = indexFilePath(ctx.workspaceRoot)
2283
+ try {
2284
+ await fsp.access(indexFile)
2285
+ } catch {
2286
+ return err(
2287
+ `codebase_search: no codebase index found at '${indexFile}'. ` +
2288
+ `The codebase has not been indexed yet. Ask a human to run ` +
2289
+ `'headlesscode index --workspace ${ctx.workspaceRoot}' first (this builds the semantic index ` +
2290
+ `and costs a small amount of embedding API money).`,
2291
+ )
2292
+ }
2293
+
2294
+ // The query embedder must match the backend the index was built with.
2295
+ // Local and cloud embedding models produce different-dimension vectors
2296
+ // (ollama qwen3-embedding:8b = 4096, openrouter qwen/qwen3-embedding-4b
2297
+ // = 2560), so a mismatch would silently compare apples to oranges.
2298
+ // The env var/flag selection happens at index-build time; the index
2299
+ // metadata records which backend built it, and we use that backend
2300
+ // here — refusing explicitly when it can't be honored.
2301
+ const metadata = loadIndexMetadata(ctx.workspaceRoot)
2302
+ let queryBackend = resolveEmbeddingBackend(process.env)
2303
+ if (metadata && metadata.backend !== queryBackend) {
2304
+ return err(
2305
+ `codebase_search: index was built with backend "${metadata.backend}" (model "${metadata.model}"), ` +
2306
+ `but ${EMBEDDING_BACKEND_ENV} selects "${queryBackend}". Embedding backends produce ` +
2307
+ `different-dimension vectors and are NOT interchangeable — rebuild the index with ` +
2308
+ `'headlesscode index --embedding-backend ${queryBackend}' or unset ` +
2309
+ `${EMBEDDING_BACKEND_ENV} to search with the backend that built the index.`,
2310
+ )
2311
+ }
2312
+ if (metadata) {
2313
+ queryBackend = metadata.backend
2314
+ }
2315
+
2316
+ // Embed the query (one call). This is a real network call; failures
2317
+ // surface as a tool error so the loop's mistake handling applies.
2318
+ const embedder = createEmbedder(queryBackend)
2319
+ const queryResult = await embedder.embedBatch([query])
2320
+ if (queryResult.embeddings.length !== 1) {
2321
+ return err(`codebase_search: embedder returned ${queryResult.embeddings.length} embeddings for the query`)
2322
+ }
2323
+
2324
+ const results = searchIndex(ctx.workspaceRoot, queryResult.embeddings[0]!, pathPrefix)
2325
+ return ok(formatSearchResults(query, results))
2326
+ })()
2327
+ }
2328
+
2329
+ function errorMessage(error: unknown): string {
2330
+ return error instanceof Error ? error.message : String(error)
2331
+ }
2332
+
2333
+ /**
2334
+ * Lazily construct the session's local summarizer. Returns undefined when the
2335
+ * feature is off (never construct an Ollama client for a non-opted-in
2336
+ * session). One executor = one summarizer, so a session that keeps producing
2337
+ * oversized results reuses the same client across calls.
2338
+ */
2339
+ function makeSummarizer(): OllamaOutputSummarizer | undefined {
2340
+ if (!isLocalSummarizationEnabled()) {
2341
+ return undefined
2342
+ }
2343
+ if (toolSummarizer === undefined) {
2344
+ toolSummarizer = new OllamaOutputSummarizer()
2345
+ }
2346
+ return toolSummarizer
2347
+ }
2348
+
2349
+ // ─── Registry ────────────────────────────────────────────────────────────────
2350
+
2351
+ /** All tool names the vendored getNativeTools() may expose, for stubbing. */
2352
+ const VENDORED_TOOL_NAMES = [
2353
+ "access_mcp_resource",
2354
+ "apply_diff",
2355
+ "apply_patch",
2356
+ "ask_followup_question",
2357
+ "attempt_completion",
2358
+ "codebase_search",
2359
+ "execute_command",
2360
+ "generate_image",
2361
+ "list_files",
2362
+ "new_task",
2363
+ "read_command_output",
2364
+ "read_file",
2365
+ "run_slash_command",
2366
+ "skill",
2367
+ "search_replace",
2368
+ "edit_file",
2369
+ "edit",
2370
+ "search_files",
2371
+ "switch_mode",
2372
+ "update_todo_list",
2373
+ "write_to_file",
2374
+ ] as const
2375
+
2376
+ const IMPLEMENTED_TOOLS = new Set([
2377
+ "read_file",
2378
+ "write_to_file",
2379
+ "apply_diff",
2380
+ "search_replace",
2381
+ "edit_file",
2382
+ "execute_command",
2383
+ "list_files",
2384
+ "codebase_search",
2385
+ "browser_action",
2386
+ "describe_image",
2387
+ "outline",
2388
+ "go_to_definition",
2389
+ "find_references",
2390
+ "import_graph",
2391
+ "update_todo_list",
2392
+ // Recursive task decomposition: implemented by HeadlessSession itself
2393
+ // (HeadlessSession.register registers handleNewTask on the executor —
2394
+ // see src/engine/loop.ts). The handler needs the SESSION (lineage,
2395
+ // budget, checkpoint service), so it lives there, not in this file's
2396
+ // stateless handler factory; read-only executors (reviewer/QA/local
2397
+ // explore) still stub it as "not implemented" via the vendored schema.
2398
+ "new_task",
2399
+ // switch_mode (plans/switch-mode-headless.md): implemented by
2400
+ // HeadlessSession itself for the same reason as new_task — the handler
2401
+ // needs SESSION state (current mode, transcript, mode-switch counter) and
2402
+ // shares ask_followup_question's decision-escalation marker protocol.
2403
+ // Read-only executors (reviewer/QA/local explore) still stub it via the
2404
+ // vendored schema.
2405
+ "switch_mode",
2406
+ ])
2407
+
2408
+ /**
2409
+ * Register the TS-ONLY code-intelligence tools + run_tests — CONDITIONALLY,
2410
+ * gated on the workspace actually being TS/JS (src/tools/language-detect.ts).
2411
+ * All of these are built on ts.Program/tsx: on a Python/C++/Rust workspace
2412
+ * they are dead weight (advertised on every request, costing prompt tokens,
2413
+ * silently useless if tried) — the same "match the tool list to what actually
2414
+ * works" discipline `codebase_search` already models with its broad extension
2415
+ * list. On non-TS workspaces they are NOT registered here, so both the
2416
+ * executor and the advertised tool list (loop.ts gates advertisement on
2417
+ * executor.has) omit them.
2418
+ */
2419
+ function registerTypeScriptGatedTools(
2420
+ executor: ToolExecutor,
2421
+ workspaceRoot: string,
2422
+ options: ToolExecutorOptions = {},
2423
+ ): void {
2424
+ if (!isTypeScriptWorkspace(workspaceRoot)) {
2425
+ return
2426
+ }
2427
+ executor.register("outline", outlineHandler)
2428
+ executor.register("go_to_definition", goToDefinitionHandler)
2429
+ executor.register("find_references", findReferencesHandler)
2430
+ executor.register("import_graph", importGraphHandler)
2431
+ // rename_symbol EDITS files, so it is deliberately NOT registered on the
2432
+ // read-only executors (reviewer/QA/local explore) — they must never be
2433
+ // able to modify source, and their tool lists don't advertise it either
2434
+ // (see the RENAME_SYMBOL_TOOL note in src/codeintel/tools.ts).
2435
+ executor.register("rename_symbol", renameSymbolHandler)
2436
+ // run_tests RUNS the project's test suite — an edit-loop tool, not a
2437
+ // read-only inspection, so like rename_symbol it is only registered on
2438
+ // the headless (edit-capable) executor. Reviewer/QA use execute_command
2439
+ // to run tests themselves; their tool lists don't advertise run_tests.
2440
+ executor.register("run_tests", (args, ctx) => runTestsHandler(args, ctx, options.getSessionChangedFiles))
2441
+ }
2442
+
2443
+ /**
2444
+ * Create the default headless executor for a workspace root: the 7 core tools
2445
+ * (read_file, write_to_file, apply_diff, search_replace, edit_file,
2446
+ * execute_command, list_files), attempt_completion + ask_followup_question
2447
+ * handlers, and stubs for every other vendored tool schema.
2448
+ */
2449
+ export function createHeadlessExecutor(workspaceRoot: string, options: ToolExecutorOptions = {}): ToolExecutor {
2450
+ const executor = new ToolExecutor(workspaceRoot, options)
2451
+
2452
+ executor.registerReadFile()
2453
+ executor.register("write_to_file", writeToFileHandler)
2454
+ executor.register("apply_diff", applyDiffHandler)
2455
+ executor.register("search_replace", searchReplaceHandler)
2456
+ executor.register("edit_file", editFileHandler)
2457
+ executor.register("set_indentation", setIndentationHandler)
2458
+ executor.register("execute_command", executeCommandHandler)
2459
+ executor.registerListFiles()
2460
+ executor.register("codebase_search", codebaseSearchHandler)
2461
+ executor.register("browser_action", browserActionHandler)
2462
+ executor.register("describe_image", describeImageHandler)
2463
+ registerTypeScriptGatedTools(executor, workspaceRoot, options)
2464
+ executor.registerTodoList()
2465
+
2466
+ executor.register("attempt_completion", attemptCompletionHandler)
2467
+ executor.register("ask_followup_question", askFollowupQuestionHandler)
2468
+
2469
+ for (const name of VENDORED_TOOL_NAMES) {
2470
+ if (executor.has(name) || IMPLEMENTED_TOOLS.has(name)) {
2471
+ continue
2472
+ }
2473
+ executor.register(name, stubHandler(name))
2474
+ }
2475
+
2476
+ return executor
2477
+ }
2478
+
2479
+ /**
2480
+ * Create a READ-ONLY executor for the reviewer mode (Phase 2) and the QA mode
2481
+ * (Phase 4): the non-edit discipline is enforced at the executor level —
2482
+ * `write_to_file` is NOT registered at all, so any attempt to modify files
2483
+ * fails with a clear "not implemented" error. Only read_file / list_files /
2484
+ * execute_command (used to re-run tests, git diff, `gh` commands, boot the
2485
+ * app) plus attempt_completion / ask_followup_question are available; every
2486
+ * other vendored tool is a stub.
2487
+ *
2488
+ * Phase 4 decision: a QA agent may RUN the application and test suites (it
2489
+ * needs execute_command) but must NOT be able to modify source files
2490
+ * unexpectedly — QA verifies and reports; remediation is a separate worker
2491
+ * cycle. The mode's `edit` group is therefore never honored here: write
2492
+ * tools are absent from BOTH the executor and the tool list advertised to
2493
+ * the model (see src/qa/qa.ts `qaTools()`).
2494
+ */
2495
+ function createReadCommandExecutor(workspaceRoot: string, options: ToolExecutorOptions = {}): ToolExecutor {
2496
+ const executor = new ToolExecutor(workspaceRoot, options)
2497
+
2498
+ executor.registerReadFile()
2499
+ executor.registerListFiles()
2500
+ executor.register("execute_command", executeCommandHandler)
2501
+ executor.register("browser_action", browserActionHandler)
2502
+ // The four read-only code-intelligence tools (TS-only) follow the same
2503
+ // language gate as the headless executor — a reviewer of a C++/Python
2504
+ // workspace must not be handed ts.Program tools.
2505
+ if (isTypeScriptWorkspace(workspaceRoot)) {
2506
+ executor.register("outline", outlineHandler)
2507
+ executor.register("go_to_definition", goToDefinitionHandler)
2508
+ executor.register("find_references", findReferencesHandler)
2509
+ executor.register("import_graph", importGraphHandler)
2510
+ }
2511
+
2512
+ executor.register("attempt_completion", attemptCompletionHandler)
2513
+ executor.register("ask_followup_question", askFollowupQuestionHandler)
2514
+
2515
+ for (const name of VENDORED_TOOL_NAMES) {
2516
+ if (executor.has(name) || name === "write_to_file") {
2517
+ continue
2518
+ }
2519
+ executor.register(name, stubHandler(name))
2520
+ }
2521
+
2522
+ return executor
2523
+ }
2524
+
2525
+ /** Reviewer executor (Phase 2): read + command, no write. See above. */
2526
+ export function createReadOnlyHeadlessExecutor(workspaceRoot: string, options: ToolExecutorOptions = {}): ToolExecutor {
2527
+ return createReadCommandExecutor(workspaceRoot, options)
2528
+ }
2529
+
2530
+ /**
2531
+ * QA executor (Phase 4): read + command, no write. Functionally identical to
2532
+ * the reviewer executor, but named for the QA role so the intent is explicit
2533
+ * at every call site: the QA agent can boot the app and run tests via
2534
+ * execute_command, but cannot modify source files (no write_to_file).
2535
+ */
2536
+ export function createQaHeadlessExecutor(workspaceRoot: string, options: ToolExecutorOptions = {}): ToolExecutor {
2537
+ return createReadCommandExecutor(workspaceRoot, options)
2538
+ }
2539
+
2540
+ /**
2541
+ * Create the STRICTLY read-only executor for the opt-in local exploration
2542
+ * phase (see src/engine/local-explore.ts): only read_file and list_files are
2543
+ * real tools. attempt_completion is registered but its result is intercepted
2544
+ * by the local loop as the explicit "I have enough, hand off" signal — it is
2545
+ * NOT a real task completion. Every other tool — execute_command,
2546
+ * codebase_search, every write tool, ask_followup_question, browser_action —
2547
+ * is a stub.
2548
+ *
2549
+ * Deliberately narrower than createReadOnlyHeadlessExecutor (reviewer/QA),
2550
+ * which includes execute_command: arbitrary command execution has real
2551
+ * side-effect potential even without file writes, and this phase must be
2552
+ * pure information-gathering. codebase_search IS included: its embedding
2553
+ * call is cloud-side (OpenRouter), so it touches no local VRAM — the
2554
+ * original exclusion assumed a local embedding model co-resident with the
2555
+ * exploration model, which the deployed cloud embedder makes moot (the
2556
+ * cloud model still gets full codebase_search access in its own turn).
2557
+ */
2558
+ export function createLocalExploreExecutor(workspaceRoot: string, options: ToolExecutorOptions = {}): ToolExecutor {
2559
+ const executor = new ToolExecutor(workspaceRoot, options)
2560
+
2561
+ executor.registerReadFile()
2562
+ executor.registerListFiles()
2563
+ executor.register("codebase_search", codebaseSearchHandler)
2564
+ executor.register("attempt_completion", attemptCompletionHandler)
2565
+
2566
+ for (const name of VENDORED_TOOL_NAMES) {
2567
+ if (executor.has(name)) {
2568
+ continue
2569
+ }
2570
+ executor.register(name, stubHandler(name, ["read_file", "list_files", "codebase_search", "attempt_completion"]))
2571
+ }
2572
+
2573
+ return executor
2574
+ }
2575
+