headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,569 @@
1
+ /**
2
+ * DECISION PROXY — plans/decision-proxy-agent.md.
3
+ *
4
+ * A small, OPT-IN (`HEADLESSCODE_DECISION_PROXY=1`, default OFF) stand-in for
5
+ * the human on `ask_followup_question` (src/tools/executor.ts's
6
+ * `escalateDecision`): watches a workspace for `.harness.needs-decision`,
7
+ * answers the question by writing `.harness.decision-answer` — the EXACT file
8
+ * a human writes via scripts/headlesscode-answer.sh — grounded in the
9
+ * session's ORIGINAL, VERBATIM task text, and logs every decision it saw and
10
+ * made. No change to `escalateDecision`/`ask_followup_question`: the proxy is
11
+ * a NEW writer of an existing answer file, so the existing timeout fallback
12
+ * stays untouched and remains the safety net.
13
+ *
14
+ * Three outcomes per question:
15
+ * 1. answered — the task text grounds a specific answer → write
16
+ * `[decision-proxy] <answer>` to the answer file. The worker's next poll
17
+ * picks it up and the session continues with real input.
18
+ * 2. uncertain — the model reports it cannot ground an answer (or no task
19
+ * text is available at all) → write NOTHING. `escalateDecision`'s poll
20
+ * loop keeps waiting and today's timeout → "must decide autonomously"
21
+ * fallback fires exactly as it does with no proxy running.
22
+ * 3. errored — the LLM call failed/timed out or the response was malformed
23
+ * → write NOTHING. Fail closed on parse ambiguity: a malformed response
24
+ * is never license to fabricate an answer.
25
+ *
26
+ * Fail-open by construction: a broken or abstaining proxy degrades to exactly
27
+ * today's behavior (the session's own timeout), mirroring the non-fatal idiom
28
+ * of src/engine/local-explore.ts — an experimental subsystem must never break
29
+ * a real session.
30
+ *
31
+ * Original task text resolution (per question — re-resolved so a worker that
32
+ * spawns AFTER the proxy starts is still grounded correctly):
33
+ * 1. `--task` (verbatim string);
34
+ * 2. `--task-file` (read verbatim, relative to the workspace root);
35
+ * 3. the orchestrator's group `task_file` content — read from
36
+ * <repo>/.worktrees/.orchestrator-state.json when the workspace is one
37
+ * of the round's worktrees (the task file sits in the main repo under
38
+ * plans/parallel-tasks/; see src/orchestrator/state.ts).
39
+ * None available → every question is treated as "uncertain" (never guess).
40
+ */
41
+
42
+ import * as fs from "node:fs"
43
+ import * as fsp from "node:fs/promises"
44
+ import * as path from "node:path"
45
+ import { setTimeout as sleep } from "node:timers/promises"
46
+
47
+ import { resolveModelForMode } from "../config/mode-models.js"
48
+ import { Logger } from "../engine/logger.js"
49
+ import type { ChatMessage, LlmClient, LlmResponse } from "../engine/types.js"
50
+ import { DEFAULT_MODEL } from "../llm/openrouter.js"
51
+ import { DECISION_ANSWER_FILENAME, NEEDS_DECISION_FILENAME } from "../tools/executor.js"
52
+
53
+ // ─── Configuration (env-var resolvable, mirroring local-explore.ts) ─────────
54
+
55
+ /** Gate env var — HEADLESSCODE_DECISION_PROXY=1 enables the proxy. */
56
+ export const DECISION_PROXY_ENV = "HEADLESSCODE_DECISION_PROXY"
57
+ /** mode-models.json extraKey consulted FIRST for the proxy's model (see src/config/mode-models.ts). */
58
+ export const DECISION_PROXY_MODEL_KEY = "_decision-proxy"
59
+ /** Poll interval, ms (matches escalateDecision's DEFAULT_DECISION_POLL_INTERVAL_MS). */
60
+ export const DECISION_PROXY_POLL_INTERVAL_ENV = "HEADLESSCODE_DECISION_PROXY_POLL_INTERVAL_MS"
61
+ export const DEFAULT_DECISION_PROXY_POLL_INTERVAL_MS = 5_000
62
+ /**
63
+ * Per-call LLM abort timeout, ms. A single short completion (question in,
64
+ * JSON out — no tool loop) should not take anywhere near the session's
65
+ * 300s default; 60s is generous while keeping a hung request from blocking
66
+ * the session (which is waiting on OUR answer, up to its own 30min timeout).
67
+ */
68
+ export const DECISION_PROXY_LLM_TIMEOUT_ENV = "HEADLESSCODE_DECISION_PROXY_LLM_TIMEOUT_MS"
69
+ export const DEFAULT_DECISION_PROXY_LLM_TIMEOUT_MS = 60_000
70
+ /**
71
+ * Max output tokens for the answer call. Raised twice after the live pilot:
72
+ * deepseek-v4-flash (a reasoning model) intermittently returned HTTP 200 with
73
+ * EMPTY final content — its reasoning trace can consume the whole output
74
+ * budget (DeepSeek counts reasoning toward max_tokens), leaving no room for
75
+ * the final `{"answer": ...}` / `{"uncertain": true}` payload. 4096 is still
76
+ * a single short completion (~$0.0006 worst case) and keeps the answer call
77
+ * comfortably inside the worker's decision window.
78
+ */
79
+ export const DEFAULT_DECISION_PROXY_MAX_TOKENS = 4096
80
+ /**
81
+ * Distinguishing prefix on a proxy-authored answer file — anyone reading
82
+ * harness.log / the dashboard's decision_answered event can tell a human
83
+ * never looked at this. (The worker trims the answer, so the prefix plus the
84
+ * answer text is exactly what the session model sees.)
85
+ */
86
+ export const DECISION_PROXY_ANSWER_PREFIX = "[decision-proxy] "
87
+ /** The proxy's own audit log, relative to the workspace root (.headlesscode is gitignored). */
88
+ export const DECISION_PROXY_LOG_FILE = ".headlesscode/decision-proxy.log"
89
+
90
+ // ─── Types ──────────────────────────────────────────────────────────────────
91
+
92
+ /** The `.harness.needs-decision` marker shape (see escalateDecision). */
93
+ export interface NeedsDecisionMarker {
94
+ question: string
95
+ suggestions?: string[]
96
+ askedAt?: string
97
+ }
98
+
99
+ export interface DecisionProxyOptions {
100
+ /** The worktree/workspace to watch. */
101
+ workspaceRoot: string
102
+ /** Original task text (verbatim). Falls back to --task-file / orchestrator state. */
103
+ task?: string
104
+ /** Task file path (relative to the workspace root), read verbatim. */
105
+ taskFile?: string
106
+ /** Model id (default: mode-models.json `_decision-proxy`, else mode/_default/OPENROUTER_MODEL/client default). */
107
+ model?: string
108
+ /** The LLM client (inject a fake in tests; OpenRouterClient in prod). */
109
+ llmClient: LlmClient
110
+ /** Poll interval, ms (default $HEADLESSCODE_DECISION_PROXY_POLL_INTERVAL_MS or 5000). */
111
+ pollIntervalMs?: number
112
+ /** Per-call LLM abort timeout, ms (default $HEADLESSCODE_DECISION_PROXY_LLM_TIMEOUT_MS or 60s). */
113
+ llmTimeoutMs?: number
114
+ /** Max output tokens for the answer call (default 400). */
115
+ maxTokens?: number
116
+ logger?: Logger
117
+ }
118
+
119
+ export type DecisionProxyOutcome = "answered" | "uncertain" | "errored"
120
+
121
+ export interface DecisionProxyQuestionResult {
122
+ outcome: DecisionProxyOutcome
123
+ /** Human-readable detail: the written answer, the abstention reason, or the error. */
124
+ detail?: string
125
+ /** Task text source, when one was resolved ("cli-task", "task-file:<path>", "orchestrator:<path>"). */
126
+ taskSource?: string
127
+ /** Wall-clock ms from marker read to the terminal decision. */
128
+ latencyMs: number
129
+ }
130
+
131
+ export type ProxyResponseParse =
132
+ | { kind: "answered"; answer: string }
133
+ | { kind: "uncertain" }
134
+ | { kind: "malformed"; detail: string }
135
+
136
+ export interface TaskTextResolution {
137
+ source: string
138
+ text: string
139
+ }
140
+
141
+ export class DecisionProxyError extends Error {}
142
+
143
+ // ─── Env resolution (mirrors local-explore.ts's pattern) ────────────────────
144
+
145
+ export function isDecisionProxyEnabled(env: NodeJS.ProcessEnv = process.env): boolean {
146
+ const v = env[DECISION_PROXY_ENV]
147
+ return v !== undefined && v !== "" && v !== "0" && v.toLowerCase() !== "false"
148
+ }
149
+
150
+ export function resolveDecisionProxyPollInterval(env: NodeJS.ProcessEnv = process.env): number {
151
+ return parsePositiveInt(env[DECISION_PROXY_POLL_INTERVAL_ENV], DEFAULT_DECISION_PROXY_POLL_INTERVAL_MS, DECISION_PROXY_POLL_INTERVAL_ENV)
152
+ }
153
+
154
+ export function resolveDecisionProxyLlmTimeout(env: NodeJS.ProcessEnv = process.env): number {
155
+ return parsePositiveInt(env[DECISION_PROXY_LLM_TIMEOUT_ENV], DEFAULT_DECISION_PROXY_LLM_TIMEOUT_MS, DECISION_PROXY_LLM_TIMEOUT_ENV)
156
+ }
157
+
158
+ function parsePositiveInt(raw: string | undefined, fallback: number, envName: string): number {
159
+ if (raw === undefined || raw.trim() === "") {
160
+ return fallback
161
+ }
162
+ const n = Number(raw)
163
+ if (!Number.isInteger(n) || n <= 0) {
164
+ throw new DecisionProxyError(`${envName} must be a positive integer, got "${raw}"`)
165
+ }
166
+ return n
167
+ }
168
+
169
+ /**
170
+ * Resolve the proxy's model via mode-models.json, consulting the
171
+ * `_decision-proxy` extraKey FIRST (same indirection pattern as
172
+ * `_condensation` — src/config/mode-models.ts). Falls through to the "code"
173
+ * mode entry → `_default` → OPENROUTER_MODEL → undefined (client default).
174
+ */
175
+ export function resolveDecisionProxyModel(
176
+ workspaceRoot: string,
177
+ env: NodeJS.ProcessEnv = process.env,
178
+ explicitModel?: string,
179
+ ): string | undefined {
180
+ return resolveModelForMode({
181
+ workspaceRoot,
182
+ mode: "code",
183
+ explicitModel,
184
+ extraKeys: [DECISION_PROXY_MODEL_KEY],
185
+ env,
186
+ })
187
+ }
188
+
189
+ // ─── Marker reading ─────────────────────────────────────────────────────────
190
+
191
+ export async function readNeedsDecisionMarker(workspaceRoot: string): Promise<NeedsDecisionMarker | null> {
192
+ const p = path.join(workspaceRoot, NEEDS_DECISION_FILENAME)
193
+ let raw: string
194
+ try {
195
+ raw = await fsp.readFile(p, "utf-8")
196
+ } catch {
197
+ return null
198
+ }
199
+ try {
200
+ const parsed = JSON.parse(raw) as Record<string, unknown>
201
+ if (typeof parsed.question !== "string" || parsed.question.trim() === "") {
202
+ return null
203
+ }
204
+ const marker: NeedsDecisionMarker = { question: parsed.question }
205
+ if (Array.isArray(parsed.suggestions)) {
206
+ const texts = parsed.suggestions.filter((s): s is string => typeof s === "string" && s.length > 0)
207
+ if (texts.length > 0) {
208
+ marker.suggestions = texts
209
+ }
210
+ }
211
+ if (typeof parsed.askedAt === "string") {
212
+ marker.askedAt = parsed.askedAt
213
+ }
214
+ return marker
215
+ } catch {
216
+ return null
217
+ }
218
+ }
219
+
220
+ /** Stable identity for one escalation instance (askedAt is unique per escalateDecision). */
221
+ export function markerKey(marker: NeedsDecisionMarker): string {
222
+ return `${marker.askedAt ?? ""}\u0000${marker.question}`
223
+ }
224
+
225
+ /**
226
+ * True when the marker we read is STILL the current one on disk (same
227
+ * question + askedAt). The worker deletes the marker when it consumes an
228
+ * answer AND when its wait times out; a stale answer file written after the
229
+ * worker moved on would be read instantly by the NEXT escalation (the poll
230
+ * loop does not clear a pre-existing answer file), so we never write for a
231
+ * marker that is no longer current.
232
+ */
233
+ async function markerStillCurrent(workspaceRoot: string, marker: NeedsDecisionMarker): Promise<boolean> {
234
+ const current = await readNeedsDecisionMarker(workspaceRoot)
235
+ if (current === null) {
236
+ return false
237
+ }
238
+ return current.question === marker.question && current.askedAt === marker.askedAt
239
+ }
240
+
241
+ async function safeUnlink(p: string): Promise<void> {
242
+ try {
243
+ await fsp.unlink(p)
244
+ } catch {
245
+ // Already gone / never existed — fine either way.
246
+ }
247
+ }
248
+
249
+ // ─── Original task text resolution ──────────────────────────────────────────
250
+
251
+ export function resolveTaskText(workspaceRoot: string, opts: { task?: string; taskFile?: string }): TaskTextResolution | null {
252
+ if (opts.task !== undefined && opts.task.trim() !== "") {
253
+ return { source: "cli-task", text: opts.task }
254
+ }
255
+ if (opts.taskFile !== undefined && opts.taskFile.trim() !== "") {
256
+ const p = path.resolve(workspaceRoot, opts.taskFile)
257
+ try {
258
+ const text = fs.readFileSync(p, "utf-8")
259
+ if (text.trim() !== "") {
260
+ return { source: `task-file:${p}`, text }
261
+ }
262
+ } catch {
263
+ // Fall through to the orchestrator-state lookup.
264
+ }
265
+ }
266
+ return resolveTaskTextFromOrchestratorState(workspaceRoot)
267
+ }
268
+
269
+ function resolveTaskTextFromOrchestratorState(workspaceRoot: string): TaskTextResolution | null {
270
+ const absWorkspace = path.resolve(workspaceRoot)
271
+ // A worktree lives at <repo>/.worktrees/<name>; the round's durable state
272
+ // file sits next to the worktrees dir: <repo>/.worktrees/.orchestrator-state.json.
273
+ const statePath = path.join(path.dirname(absWorkspace), ".orchestrator-state.json")
274
+ let state: unknown
275
+ try {
276
+ state = JSON.parse(fs.readFileSync(statePath, "utf-8"))
277
+ } catch {
278
+ return null
279
+ }
280
+ if (state === null || typeof state !== "object") {
281
+ return null
282
+ }
283
+ const groups = (state as Record<string, unknown>).groups
284
+ if (!Array.isArray(groups)) {
285
+ return null
286
+ }
287
+ const repoRoot = path.dirname(path.dirname(statePath))
288
+ for (const g of groups) {
289
+ if (g === null || typeof g !== "object") {
290
+ continue
291
+ }
292
+ const group = g as Record<string, unknown>
293
+ if (typeof group.worktree !== "string" || typeof group.task_file !== "string") {
294
+ continue
295
+ }
296
+ if (path.resolve(repoRoot, group.worktree) !== absWorkspace) {
297
+ continue
298
+ }
299
+ const taskPath = path.resolve(repoRoot, group.task_file)
300
+ try {
301
+ const text = fs.readFileSync(taskPath, "utf-8")
302
+ if (text.trim() !== "") {
303
+ return { source: `orchestrator:${taskPath}`, text }
304
+ }
305
+ } catch {
306
+ return null
307
+ }
308
+ }
309
+ return null
310
+ }
311
+
312
+ // ─── LLM call ───────────────────────────────────────────────────────────────
313
+
314
+ /**
315
+ * Max attempts for ONE question's LLM call. A reasoning model (e.g. the
316
+ * default deepseek/deepseek-v4-flash-0731) sometimes returns HTTP 200 with EMPTY
317
+ * content on the first completion (observed live in the decision-proxy pilot);
318
+ * the retry re-issues the SAME one-shot prompt — still a single question, no
319
+ * tools, no loop. The caller still fails open (writes nothing) if every
320
+ * attempt is unusable.
321
+ */
322
+ export const MAX_PROXY_LLM_ATTEMPTS = 2
323
+
324
+ export const DECISION_PROXY_SYSTEM_PROMPT = `You are the DECISION PROXY for a headless coding agent. The agent was given a task and, during the session, asked a question that would normally go to a human. You answer on the human's behalf — but ONLY when the agent's ORIGINAL TASK TEXT genuinely grounds a specific answer.
325
+
326
+ You will be given:
327
+ - The agent's original task text (verbatim).
328
+ - The question the agent asked.
329
+ - Optional suggested answers.
330
+
331
+ Rules:
332
+ - If the original task text gives you enough information to answer the question DIRECTLY and SPECIFICALLY, answer it. When one of the suggested answers clearly matches what the task requires, choose it. Be concise and concrete — your answer is fed back to the agent verbatim.
333
+ - If the question is a MODE-SWITCH approval (its suggested answers are exactly "approve"/"deny"), your answer must be exactly "approve" or "deny": approve only when the original task text supports the switch (e.g. the task says the work should be handed to another mode), otherwise deny.
334
+ - If the original task text does NOT answer the question (it is silent on the matter, or the question is a genuine choice the task left open), you MUST respond with the uncertain sentinel.
335
+ - NEVER guess, invent, or extrapolate beyond what the task text supports. A fabricated answer actively misdirects the agent; an abstention merely falls back to the existing no-answer behavior.
336
+
337
+ Respond with STRICT JSON ONLY, exactly one of:
338
+ {"answer": "<your answer text>"}
339
+ {"uncertain": true}
340
+
341
+ No prose outside the JSON.`
342
+
343
+ export function buildProxyUserPrompt(taskText: string, question: string, suggestions?: string[]): string {
344
+ let out = `ORIGINAL TASK TEXT (verbatim):\n${taskText}\n\nQUESTION ASKED:\n${question}`
345
+ if (suggestions && suggestions.length > 0) {
346
+ out += `\n\nSUGGESTED ANSWERS:\n${suggestions.map((s, i) => `${i + 1}. ${s}`).join("\n")}`
347
+ }
348
+ return out
349
+ }
350
+
351
+ async function callProxyLlm(options: DecisionProxyOptions, taskText: string, marker: NeedsDecisionMarker): Promise<string> {
352
+ // Concrete model always: _decision-proxy key → mode/_default/OPENROUTER_MODEL
353
+ // → the client's own DEFAULT_MODEL (same final fallback as the main loop).
354
+ const model = options.model ?? resolveDecisionProxyModel(options.workspaceRoot, process.env) ?? DEFAULT_MODEL
355
+ const timeoutMs = options.llmTimeoutMs ?? resolveDecisionProxyLlmTimeout(process.env)
356
+ const messages: ChatMessage[] = [
357
+ { role: "system", content: DECISION_PROXY_SYSTEM_PROMPT },
358
+ { role: "user", content: buildProxyUserPrompt(taskText, marker.question, marker.suggestions) },
359
+ ]
360
+ let lastError: unknown
361
+ for (let attempt = 1; attempt <= MAX_PROXY_LLM_ATTEMPTS; attempt++) {
362
+ const controller = new AbortController()
363
+ const timer = setTimeout(() => controller.abort(), timeoutMs)
364
+ try {
365
+ const response: LlmResponse = await options.llmClient.createChatCompletion({
366
+ model,
367
+ messages,
368
+ temperature: 0,
369
+ maxTokens: options.maxTokens ?? DEFAULT_DECISION_PROXY_MAX_TOKENS,
370
+ signal: controller.signal,
371
+ })
372
+ const content = response.message.content
373
+ if (typeof content === "string" && content.trim() !== "") {
374
+ return content
375
+ }
376
+ // Diagnose the empty response for the audit trail: was the completion
377
+ // cut mid-reasoning (DeepSeek counts reasoning toward max_tokens), or
378
+ // did the model emit a reasoning block with no final content at all?
379
+ const reasoning = typeof response.message.reasoning === "string" && response.message.reasoning.length > 0
380
+ lastError = new DecisionProxyError(
381
+ `empty LLM response (attempt ${attempt}/${MAX_PROXY_LLM_ATTEMPTS})` +
382
+ (reasoning ? ` — model emitted ${response.message.reasoning!.length} chars of reasoning but no final content` : " — no content and no reasoning"),
383
+ )
384
+ } catch (err) {
385
+ lastError = err
386
+ } finally {
387
+ clearTimeout(timer)
388
+ }
389
+ }
390
+ throw lastError instanceof Error ? lastError : new DecisionProxyError(String(lastError))
391
+ }
392
+
393
+ // ─── Sentinel parsing ───────────────────────────────────────────────────────
394
+
395
+ /**
396
+ * Parse the proxy LLM's response. STRICT by design: `{"uncertain": true}`
397
+ * abstains, `{"answer": "<non-empty>"}` answers, and ANYTHING else is
398
+ * malformed — the caller treats malformed exactly like uncertain (write
399
+ * nothing) but logs it separately so a broken proxy/model is visible in the
400
+ * audit trail instead of masquerading as a legitimate abstention.
401
+ */
402
+ export function parseProxyResponse(content: string): ProxyResponseParse {
403
+ const trimmed = content.trim()
404
+ if (trimmed === "") {
405
+ return { kind: "malformed", detail: "empty response" }
406
+ }
407
+ const parsed = tryParseJson(trimmed)
408
+ if (parsed === undefined) {
409
+ return { kind: "malformed", detail: "response is not valid JSON" }
410
+ }
411
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
412
+ return { kind: "malformed", detail: "response is not a JSON object" }
413
+ }
414
+ const obj = parsed as Record<string, unknown>
415
+ if (obj.uncertain === true) {
416
+ return { kind: "uncertain" }
417
+ }
418
+ if (typeof obj.answer === "string" && obj.answer.trim() !== "") {
419
+ return { kind: "answered", answer: obj.answer.trim() }
420
+ }
421
+ return { kind: "malformed", detail: "response has neither a non-empty string 'answer' nor 'uncertain': true" }
422
+ }
423
+
424
+ /** JSON.parse with a single markdown-fence retry (models sometimes wrap JSON). */
425
+ function tryParseJson(text: string): unknown | undefined {
426
+ try {
427
+ return JSON.parse(text)
428
+ } catch {
429
+ const fenceMatch = /^```(?:json)?\s*\n([\s\S]*?)\n```\s*$/.exec(text)
430
+ if (fenceMatch) {
431
+ try {
432
+ return JSON.parse(fenceMatch[1])
433
+ } catch {
434
+ return undefined
435
+ }
436
+ }
437
+ return undefined
438
+ }
439
+ }
440
+
441
+ // ─── One question ───────────────────────────────────────────────────────────
442
+
443
+ export async function processQuestion(
444
+ options: DecisionProxyOptions,
445
+ marker: NeedsDecisionMarker,
446
+ ): Promise<DecisionProxyQuestionResult> {
447
+ const startedAt = Date.now()
448
+ const answerPath = path.join(options.workspaceRoot, DECISION_ANSWER_FILENAME)
449
+
450
+ // Re-resolved per question (cheap file reads) so a worker that spawns after
451
+ // the proxy starts — or an orchestrator state file written late — is still
452
+ // grounded correctly.
453
+ const task = resolveTaskText(options.workspaceRoot, { task: options.task, taskFile: options.taskFile })
454
+ if (task === null) {
455
+ return { outcome: "uncertain", detail: "no original task text available — cannot ground an answer", latencyMs: Date.now() - startedAt }
456
+ }
457
+
458
+ let content: string
459
+ try {
460
+ content = await callProxyLlm(options, task.text, marker)
461
+ } catch (err) {
462
+ return { outcome: "errored", detail: errorMessage(err), latencyMs: Date.now() - startedAt }
463
+ }
464
+
465
+ const parsed = parseProxyResponse(content)
466
+ if (parsed.kind === "uncertain") {
467
+ return { outcome: "uncertain", detail: "model reported uncertain — task text cannot ground an answer", taskSource: task.source, latencyMs: Date.now() - startedAt }
468
+ }
469
+ if (parsed.kind === "malformed") {
470
+ return { outcome: "errored", detail: `malformed proxy response: ${parsed.detail}`, taskSource: task.source, latencyMs: Date.now() - startedAt }
471
+ }
472
+
473
+ // Pre-write guard: only answer the question that is STILL the current one.
474
+ // If the marker disappeared while we were thinking, the worker already
475
+ // moved on (answered or timed out) — writing now could poison the NEXT
476
+ // escalation with a stale answer (escalateDecision reads an existing
477
+ // answer file instantly and never clears a stale one).
478
+ if (!(await markerStillCurrent(options.workspaceRoot, marker))) {
479
+ return { outcome: "errored", detail: "marker disappeared before the answer could be written — skipped", taskSource: task.source, latencyMs: Date.now() - startedAt }
480
+ }
481
+
482
+ await fsp.writeFile(answerPath, `${DECISION_PROXY_ANSWER_PREFIX}${parsed.answer}`, "utf-8")
483
+
484
+ // Post-write guard: if the marker is gone/changed right after the write,
485
+ // remove the answer we just wrote. Either the worker already consumed it
486
+ // (its read happens before its marker unlink — deleting the file is
487
+ // harmless) or the worker timed out at the same instant (the file is
488
+ // stale and would poison the next escalation — deleting is REQUIRED).
489
+ if (!(await markerStillCurrent(options.workspaceRoot, marker))) {
490
+ await safeUnlink(answerPath)
491
+ return { outcome: "answered", detail: `${parsed.answer} (answer removed post-write — marker already gone, consumed or stale)`, taskSource: task.source, latencyMs: Date.now() - startedAt }
492
+ }
493
+
494
+ return { outcome: "answered", detail: parsed.answer, taskSource: task.source, latencyMs: Date.now() - startedAt }
495
+ }
496
+
497
+ // ─── The poll loop ──────────────────────────────────────────────────────────
498
+
499
+ /**
500
+ * Run the decision proxy until `signal` aborts (SIGINT/SIGTERM in the CLI;
501
+ * an injected controller in tests). Never throws out of the loop: a question
502
+ * that fails to process is logged and the loop keeps watching.
503
+ */
504
+ export async function runDecisionProxy(options: DecisionProxyOptions, signal?: AbortSignal): Promise<void> {
505
+ const logger = options.logger ?? new Logger()
506
+ const pollIntervalMs = options.pollIntervalMs ?? resolveDecisionProxyPollInterval(process.env)
507
+ const resolvedModel = options.model ?? resolveDecisionProxyModel(options.workspaceRoot, process.env)
508
+
509
+ logger.info("decision-proxy started", {
510
+ workspaceRoot: options.workspaceRoot,
511
+ model: resolvedModel ?? "(client default)",
512
+ pollIntervalMs,
513
+ llmTimeoutMs: options.llmTimeoutMs ?? resolveDecisionProxyLlmTimeout(process.env),
514
+ })
515
+
516
+ // The last escalation instance handled. The marker STAYS while the worker
517
+ // waits (up to its decision timeout), so without this we would re-call the
518
+ // LLM on the same question every poll. A new escalation always has a fresh
519
+ // askedAt, so the key never collides across questions.
520
+ let lastHandledKey: string | undefined
521
+
522
+ for (;;) {
523
+ if (signal?.aborted) {
524
+ logger.info("decision-proxy stopped")
525
+ return
526
+ }
527
+ const marker = await readNeedsDecisionMarker(options.workspaceRoot)
528
+ if (marker !== null) {
529
+ const key = markerKey(marker)
530
+ if (key !== lastHandledKey) {
531
+ lastHandledKey = key
532
+ let result: DecisionProxyQuestionResult
533
+ try {
534
+ result = await processQuestion({ ...options, logger }, marker)
535
+ } catch (err) {
536
+ // Never let a processing failure reject the loop — fail open
537
+ // to today's behavior (the worker's own timeout).
538
+ logger.error("decision-proxy question — ERRORED, wrote nothing", {
539
+ askedAt: marker.askedAt,
540
+ question: marker.question,
541
+ outcome: "errored",
542
+ detail: errorMessage(err),
543
+ })
544
+ continue
545
+ }
546
+ const meta = {
547
+ askedAt: marker.askedAt,
548
+ question: marker.question,
549
+ outcome: result.outcome,
550
+ latencyMs: result.latencyMs,
551
+ ...(result.taskSource ? { taskSource: result.taskSource } : {}),
552
+ ...(result.detail ? { detail: result.detail } : {}),
553
+ }
554
+ if (result.outcome === "answered") {
555
+ logger.info("decision-proxy question", meta)
556
+ } else if (result.outcome === "uncertain") {
557
+ logger.warn("decision-proxy question — UNCERTAIN, wrote nothing", meta)
558
+ } else {
559
+ logger.error("decision-proxy question — ERRORED, wrote nothing", meta)
560
+ }
561
+ }
562
+ }
563
+ await sleep(pollIntervalMs)
564
+ }
565
+ }
566
+
567
+ function errorMessage(error: unknown): string {
568
+ return error instanceof Error ? error.message : String(error)
569
+ }