headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,291 @@
1
+ /**
2
+ * Shared types for the headless harness engine.
3
+ *
4
+ * These are deliberately minimal, local interfaces for the OpenAI-compatible
5
+ * chat message shapes used by OpenRouter. We do NOT pull in the `openai` SDK —
6
+ * the vendored Zoo Code core already provides its own type-only shim for the
7
+ * tool schemas (see `src/vendor/zoo-code/shim/openai.d.ts`), and the harness
8
+ * only needs the handful of message shapes below to talk to a chat-completions
9
+ * endpoint.
10
+ */
11
+
12
+ import type { PermissionsConfig } from "../permissions/config.js"
13
+
14
+ export type ChatRole = "system" | "user" | "assistant" | "tool"
15
+
16
+ /** A single OpenAI-style function call emitted by the model. */
17
+ export interface ChatToolCall {
18
+ id: string
19
+ type: "function"
20
+ function: {
21
+ name: string
22
+ arguments: string
23
+ }
24
+ }
25
+
26
+ /** A chat message in OpenAI/OpenRouter chat-completions format. */
27
+ export interface ChatMessage {
28
+ role: ChatRole
29
+ content: string | null
30
+ tool_calls?: ChatToolCall[]
31
+ /** Present on `role: "tool"` messages, links back to the assistant call. */
32
+ tool_call_id?: string
33
+ /** Present on `role: "tool"` messages. */
34
+ name?: string
35
+ /**
36
+ * The model's reasoning/"thinking" text for this assistant message
37
+ * (streaming-and-reasoning). OpenRouter normalizes DeepSeek's
38
+ * `reasoning_content` to `reasoning` on the response message; we echo it
39
+ * back onto outgoing assistant history the same way. Optional — absent
40
+ * for models/calls that don't produce reasoning content.
41
+ */
42
+ reasoning?: string
43
+ }
44
+
45
+ /** An OpenAI-format ChatCompletionTool (function schema). */
46
+ export interface ChatTool {
47
+ type: "function"
48
+ function: {
49
+ name: string
50
+ description?: string
51
+ strict?: boolean | null
52
+ parameters?: Record<string, unknown>
53
+ [key: string]: unknown
54
+ }
55
+ [key: string]: unknown
56
+ }
57
+
58
+ /** Request shape accepted by an LlmClient. */
59
+ export interface LlmRequest {
60
+ model: string
61
+ messages: ChatMessage[]
62
+ tools?: ChatTool[]
63
+ temperature?: number
64
+ maxTokens?: number
65
+ signal?: AbortSignal
66
+ /**
67
+ * Opt-in SSE streaming (streaming-and-reasoning). When true the client
68
+ * streams deltas and assembles the final message from the stream instead
69
+ * of one blocking fetch. Default OFF — existing callers/tests are
70
+ * unaffected.
71
+ */
72
+ stream?: boolean
73
+ /**
74
+ * Streaming-and-reasoning: invoked for each incremental chunk as a streamed
75
+ * response arrives (kind "text" | "reasoning" | "tool"). Only called when
76
+ * `stream` is true and the client actually streams; the loop uses it to emit
77
+ * `llm_stream_chunk` events for the dashboard's live-typing view. Non-fatal
78
+ * by contract: a throw from this callback must not fail the LLM call.
79
+ */
80
+ onStreamChunk?: (kind: "text" | "reasoning" | "tool", chunk: string) => void
81
+ /**
82
+ * Graded reasoning effort for models that support it (the pinned DeepSeek
83
+ * family — issue #30 experiment). User-facing vocabulary is DeepSeek's
84
+ * native set low/medium/high/max, plus OpenRouter's normalized "xhigh"
85
+ * (an alias for the native max); the client normalizes to OpenRouter's
86
+ * `reasoning: { effort }` field on the wire. Optional — when unset, no
87
+ * effort field is sent and the endpoint's own undeclared default applies
88
+ * (pre-existing behavior, unchanged).
89
+ */
90
+ reasoningEffort?: string
91
+ /**
92
+ * Sampling-level override for llama.cpp's `repeat_penalty`, applied for
93
+ * exactly one request. The loop sets this once its identical-consecutive-
94
+ * call guardrail (see DEFAULT_IDENTICAL_CALL_NUDGE_THRESHOLD in
95
+ * src/engine/loop.ts) detects a repeat starting, as an alternative to a
96
+ * text-only nudge — three rounds of increasingly specific injected
97
+ * corrections were verified live not to reliably interrupt a local model
98
+ * mid-repetition (2026-08-21/22), so this escalates at the sampler
99
+ * instead of only in the prompt. Ignored by clients/backends that don't
100
+ * support per-request sampling overrides (e.g. OpenRouter).
101
+ */
102
+ repeatPenalty?: number
103
+ /**
104
+ * Forces the model to call SOME tool rather than allowing a free
105
+ * text/empty response — the OpenAI-style `"required"` value (also
106
+ * `"none"`/`"auto"`). Verified live 2026-08-20: once the identical-call
107
+ * guard's tool exclusion removes a strongly-preferred tool from the
108
+ * schema, the model doesn't substitute a different tool call — it
109
+ * produces a genuinely EMPTY (zero-token) generation instead, 100% of
110
+ * the time (148/148 in one trial), consistent with the default
111
+ * `tool_choice: "auto"` leaving a free-text/empty branch available for
112
+ * the grammar-constrained decoder to collapse into once its preferred
113
+ * path is removed. `"required"` closes that branch. Ignored by
114
+ * clients/backends that don't support it.
115
+ */
116
+ toolChoice?: "auto" | "none" | "required"
117
+ }
118
+
119
+ /** Response shape returned by an LlmClient. */
120
+ export interface LlmResponse {
121
+ message: ChatMessage
122
+ usage?: {
123
+ promptTokens?: number
124
+ completionTokens?: number
125
+ totalTokens?: number
126
+ /**
127
+ * Prompt tokens served from the provider's prefix cache (a subset of
128
+ * promptTokens, not additional) — OpenRouter surfaces this as
129
+ * `usage.prompt_tokens_details.cached_tokens` when the underlying
130
+ * provider (e.g. DeepSeek) supports automatic prompt caching. Cached
131
+ * tokens are billed at a steep discount; see src/budget/cost.ts.
132
+ */
133
+ cachedTokens?: number
134
+ }
135
+ }
136
+
137
+ /**
138
+ * The LLM client contract the orchestration loop depends on.
139
+ *
140
+ * The loop must NOT hardcode OpenRouter: tests inject a fake client that
141
+ * implements this interface, so the loop is fully unit-testable without a
142
+ * network or API key.
143
+ */
144
+ export interface LlmClient {
145
+ createChatCompletion(request: LlmRequest): Promise<LlmResponse>
146
+ }
147
+
148
+ /** Result of executing one tool call. */
149
+ export interface ToolResult {
150
+ content: string
151
+ isError: boolean
152
+ }
153
+
154
+ /**
155
+ * Usage of one LLM call made OUTSIDE the main loop's request path — e.g. the
156
+ * cloud vision captioning in src/vision/describe.ts (browser screenshots and
157
+ * the `describe_image` tool). Auxiliary calls are recorded into the SAME
158
+ * BudgetTracker + running session totals as a main call (see
159
+ * recordAuxLlmUsage in src/engine/loop.ts), so their real token/cost shows up
160
+ * in the session's budget/usage accounting instead of being an untracked side
161
+ * channel. Shape mirrors the token counts of LlmResponse.usage.
162
+ */
163
+ export interface AuxLlmUsage {
164
+ /** The model id the auxiliary call actually ran (echoed by the provider). */
165
+ model: string
166
+ inputTokens: number
167
+ outputTokens: number
168
+ /** Subset of inputTokens served from the provider's prompt cache. */
169
+ cachedTokens?: number
170
+ }
171
+
172
+ /** Per-call context handed to tool handlers. */
173
+ export interface ToolContext {
174
+ workspaceRoot: string
175
+ /**
176
+ * Resolved command allow/deny + protected-file permissions for this
177
+ * executor/session (see src/permissions/). Always present — the executor
178
+ * resolves built-in defaults when nothing is configured — so handlers can
179
+ * enforce command gating (execute_command) and protected-file refusals
180
+ * (write_to_file) without guessing.
181
+ */
182
+ permissions: PermissionsConfig
183
+ /**
184
+ * See HeadlessSessionConfig.guardLargeOverwrites (loop.ts) for the full
185
+ * writeup. When true, `write_to_file` refuses to overwrite an existing
186
+ * file that already has substantial content (see the guard in
187
+ * `writeToFileHandler`, src/tools/executor.ts) — creating a brand-new
188
+ * file is never affected. Absent/false for cloud sessions and bare
189
+ * executors (tests, reviewer/QA).
190
+ */
191
+ guardLargeOverwrites?: boolean
192
+ /**
193
+ * Decision escalation (ask_followup_question, see src/tools/executor.ts):
194
+ * how long to block waiting for `.harness.decision-answer` before falling
195
+ * back to today's autonomous-decision error, and how often to poll for it.
196
+ * Both optional — handlers fall back to their own defaults when absent.
197
+ */
198
+ decisionTimeoutMs?: number
199
+ decisionPollIntervalMs?: number
200
+ /**
201
+ * Budget-clock pause/resume hooks, wired by HeadlessSession right after it
202
+ * constructs a BudgetTracker (see src/budget/budget.ts's pauseClock /
203
+ * resumeClock). ask_followup_question calls these around its blocking wait
204
+ * so time spent waiting on a human/orchestrator answer isn't charged
205
+ * against the session's duration budget. Absent when no budget is
206
+ * configured, or when the injected executor never wired them.
207
+ */
208
+ pauseBudgetClock?: () => void
209
+ resumeBudgetClock?: () => void
210
+ /**
211
+ * Live worker monitoring: fired at the same lifecycle points where the
212
+ * `.harness.needs-decision` marker is written/cleared (see the
213
+ * ask_followup_question handler in src/tools/executor.ts), so the session
214
+ * can mirror those transitions on its structured event feed
215
+ * (`decision_blocked` / `decision_answered`). Absent when no hook was
216
+ * wired (plain executor use, tests without a session).
217
+ */
218
+ onDecisionEvent?: (eventType: "decision_blocked" | "decision_answered", fields: Record<string, unknown>) => void
219
+ /**
220
+ * Live todo-list monitoring: fired each time update_todo_list replaces the
221
+ * session's checklist (see the handler in src/tools/executor.ts), carrying
222
+ * the full normalized checklist plus done/in-progress/pending counts, so
223
+ * the session can mirror it on its structured event feed (`todo_updated`).
224
+ * Absent when no hook was wired (plain executor use, tests without a
225
+ * session).
226
+ */
227
+ onTodoEvent?: (fields: { todos: string; done: number; inProgress: number; pending: number }) => void
228
+ /**
229
+ * Auxiliary LLM usage reporting (cloud vision captioning — see
230
+ * src/vision/describe.ts): fired after every LLM call made outside the
231
+ * main loop's request path, so the session can record its tokens/cost into
232
+ * the same BudgetTracker + totals as a main call (see recordAuxLlmUsage in
233
+ * src/engine/loop.ts). Absent when no hook was wired — plain executor use,
234
+ * tests without a session. Its presence also gates the screenshot action's
235
+ * auto-describe behavior (browser_action only captions screenshots when
236
+ * attached to a real accounting session; bare executors skip the call and
237
+ * the model can still use `describe_image` explicitly).
238
+ */
239
+ onAuxLlmUsage?: (usage: AuxLlmUsage) => void
240
+ }
241
+
242
+ /** A tool handler: dispatch any registered tool by name. */
243
+ export type ToolHandler = (args: Record<string, unknown>, ctx: ToolContext) => Promise<ToolResult> | ToolResult
244
+
245
+ /** Parsed result of one assistant tool call. */
246
+ export interface ParsedToolCall {
247
+ id: string
248
+ name: string
249
+ args: Record<string, unknown>
250
+ rawArguments: string
251
+ /** Set when JSON.parse failed and best-effort extraction also failed. */
252
+ parseError?: string
253
+ }
254
+
255
+ export type SessionStatus = "success" | "error"
256
+
257
+ /**
258
+ * Phase 6 — budget accounting surfaced on every SessionResult when a
259
+ * per-session budget is configured (null budget → absent → zero change).
260
+ */
261
+ export interface SessionBudgetUsage {
262
+ /** Estimated USD spend (tokens × pricing) accumulated across LLM calls. */
263
+ costUsd: number
264
+ /** Wall-clock elapsed since the session's budget tracker started, ms. */
265
+ elapsedMs: number
266
+ /** Number of LLM calls (ticks) performed. */
267
+ iterations: number
268
+ /** Model id the session ran with. */
269
+ model: string
270
+ }
271
+
272
+ export interface SessionResult {
273
+ status: SessionStatus
274
+ result?: string
275
+ error?: string
276
+ /** Machine-readable failure reason (e.g. "budget") for callers/CLI. */
277
+ reason?: string
278
+ iterations: number
279
+ toolCalls: number
280
+ /**
281
+ * Issue #34: absolute path to the session's complete final report
282
+ * (`<workspaceRoot>/.headlesscode/reports/<sessionId>.md`), present when
283
+ * the session ended successfully (attempt_completion or the text-only
284
+ * fallback) and the report write succeeded. Callers (runQa/runReview →
285
+ * orchestrator state) persist this so the full reasoning behind a
286
+ * review/QA verdict is one file-read away, not a re-run away.
287
+ */
288
+ reportPath?: string
289
+ /** Phase 6: present when the session had a budget (see SessionBudgetUsage). */
290
+ budgetUsage?: SessionBudgetUsage
291
+ }
@@ -0,0 +1,186 @@
1
+ /**
2
+ * Cost/token monitoring (workstream 3) — per-session usage persistence.
3
+ *
4
+ * Storage layout, one JSONL file per session, append-only, mirroring the
5
+ * existing project idiom (`src/memory/local.ts`'s `appendJsonl`/`readJsonl`
6
+ * — plain `node:fs`, no schema library, loose validation on read):
7
+ *
8
+ * <workspaceRoot>/.headlesscode/usage/<sessionId>.jsonl
9
+ *
10
+ * A worker process normally writes exactly ONE line here (its single
11
+ * completed session), but the format is append-only JSONL rather than a
12
+ * single JSON object so that concurrent/rerun scenarios (e.g. a worktree
13
+ * reused across multiple sessions) never corrupt a partial write — same
14
+ * write-contention rationale as the memory store's per-project files.
15
+ *
16
+ * This module is intentionally the ONLY place that knows the on-disk usage
17
+ * layout, so `src/dashboard/aggregate.ts` (read side) and
18
+ * `src/engine/loop.ts` (write side) stay in sync.
19
+ */
20
+
21
+ import * as fsp from "node:fs/promises"
22
+ import * as path from "node:path"
23
+
24
+ /** One completed session's usage, as persisted to `.headlesscode/usage/<sessionId>.jsonl`. */
25
+ export interface UsageRecord {
26
+ sessionId: string
27
+ mode: string
28
+ model: string
29
+ iterations: number
30
+ inputTokens: number
31
+ outputTokens: number
32
+ /**
33
+ * Subset of inputTokens served from the provider's prompt cache (see
34
+ * LlmResponse.usage.cachedTokens). Optional: absent on records written
35
+ * before this field existed, or when the provider never reports it.
36
+ */
37
+ cachedTokens?: number
38
+ costUsd: number
39
+ startedAt: string
40
+ endedAt: string
41
+ status: "success" | "error" | "budget"
42
+ workspaceRoot: string
43
+ }
44
+
45
+ /**
46
+ * A live (in-progress) usage snapshot, overwritten in place at
47
+ * `.headlesscode/usage/<sessionId>.live.json` after every iteration while a
48
+ * session runs, so the dashboard can show accumulating cost/tokens before the
49
+ * session finishes. Same shape as `UsageRecord` minus `endedAt`, with
50
+ * `status: "running"`.
51
+ */
52
+ export interface LiveUsageRecord {
53
+ sessionId: string
54
+ mode: string
55
+ model: string
56
+ iterations: number
57
+ inputTokens: number
58
+ outputTokens: number
59
+ cachedTokens?: number
60
+ costUsd: number
61
+ startedAt: string
62
+ status: "running"
63
+ workspaceRoot: string
64
+ }
65
+
66
+ /** The usage dir for a workspace: `<workspaceRoot>/.headlesscode/usage`. */
67
+ export function usageDir(workspaceRoot: string): string {
68
+ return path.join(workspaceRoot, ".headlesscode", "usage")
69
+ }
70
+
71
+ /** The usage file path for one session. */
72
+ export function usageFilePath(workspaceRoot: string, sessionId: string): string {
73
+ return path.join(usageDir(workspaceRoot), `${sessionId}.jsonl`)
74
+ }
75
+
76
+ /** The live (in-progress) usage snapshot path for one session. */
77
+ export function liveUsageFilePath(workspaceRoot: string, sessionId: string): string {
78
+ return path.join(usageDir(workspaceRoot), `${sessionId}.live.json`)
79
+ }
80
+
81
+ /**
82
+ * Append one usage record for a completed session. Creates the usage dir if
83
+ * needed. Callers (see `HeadlessSession.recordUsage`) are expected to wrap
84
+ * this in try/catch and treat failures as non-fatal — this function itself
85
+ * does not swallow errors, so it fails loudly for direct callers/tests.
86
+ */
87
+ export async function recordSessionUsage(workspaceRoot: string, record: UsageRecord): Promise<void> {
88
+ const file = usageFilePath(workspaceRoot, record.sessionId)
89
+ await fsp.mkdir(path.dirname(file), { recursive: true })
90
+ await fsp.appendFile(file, JSON.stringify(record) + "\n", "utf-8")
91
+ }
92
+
93
+ /**
94
+ * Overwrite a session's live (in-progress) usage snapshot. Never appended —
95
+ * this is a point-in-time snapshot, not a log (unlike `recordSessionUsage`).
96
+ * Creates the usage dir if needed. Callers (see `HeadlessSession`) are
97
+ * expected to wrap this in try/catch and treat failures as non-fatal.
98
+ */
99
+ export async function writeLiveUsage(workspaceRoot: string, record: LiveUsageRecord): Promise<void> {
100
+ const file = liveUsageFilePath(workspaceRoot, record.sessionId)
101
+ await fsp.mkdir(path.dirname(file), { recursive: true })
102
+ await fsp.writeFile(file, JSON.stringify(record, null, 2) + "\n", "utf-8")
103
+ }
104
+
105
+ /**
106
+ * Delete a session's live snapshot. Called on every completion path once the
107
+ * final `.jsonl` record is written — that record is the authoritative source,
108
+ * so a stale live snapshot must not linger. Never throws for a missing file
109
+ * (`force: true`). Callers are expected to wrap this in try/catch (non-fatal).
110
+ */
111
+ export async function removeLiveUsage(workspaceRoot: string, sessionId: string): Promise<void> {
112
+ await fsp.rm(liveUsageFilePath(workspaceRoot, sessionId), { force: true })
113
+ }
114
+
115
+ /**
116
+ * Read + loosely validate one usage JSONL file. Malformed/partially-written
117
+ * lines are skipped; a missing file yields an empty array (never throws for
118
+ * ENOENT — matches the memory store's `readJsonl` idiom).
119
+ *
120
+ * Deduped by `sessionId` (issue #81): a session id can be reused across a
121
+ * restart/retry of the same worker, which appends another line to the same
122
+ * file (the file itself is named `<sessionId>.jsonl`, so every line here
123
+ * already shares one id) — without dedup, dashboard `sumSessions` would sum
124
+ * every one of those rows and double/triple-count cost and tokens. The last
125
+ * record for a given sessionId wins (most recent write reflects the final
126
+ * outcome of that session id).
127
+ */
128
+ export async function readUsageFile(file: string): Promise<UsageRecord[]> {
129
+ let raw: string
130
+ try {
131
+ raw = await fsp.readFile(file, "utf-8")
132
+ } catch (error) {
133
+ if ((error as NodeJS.ErrnoException).code === "ENOENT") {
134
+ return []
135
+ }
136
+ throw error
137
+ }
138
+ const bySessionId = new Map<string, UsageRecord>()
139
+ const order: string[] = []
140
+ for (const line of raw.split("\n")) {
141
+ const trimmed = line.trim()
142
+ if (!trimmed) {
143
+ continue
144
+ }
145
+ try {
146
+ const parsed = JSON.parse(trimmed) as Partial<UsageRecord>
147
+ if (typeof parsed.sessionId === "string" && typeof parsed.status === "string") {
148
+ if (!bySessionId.has(parsed.sessionId)) {
149
+ order.push(parsed.sessionId)
150
+ }
151
+ bySessionId.set(parsed.sessionId, parsed as UsageRecord)
152
+ }
153
+ } catch {
154
+ // Loose validation: skip malformed / partially-written lines.
155
+ }
156
+ }
157
+ return order.map((id) => bySessionId.get(id) as UsageRecord)
158
+ }
159
+
160
+ /**
161
+ * Read + loosely validate one live usage snapshot (`*.live.json` — a single
162
+ * JSON object, overwritten in place, NOT JSONL). Missing or malformed files
163
+ * yield `null` (never throws for ENOENT / bad JSON — same ethos as
164
+ * `readUsageFile`).
165
+ */
166
+ export async function readLiveUsageFile(file: string): Promise<LiveUsageRecord | null> {
167
+ let raw: string
168
+ try {
169
+ raw = await fsp.readFile(file, "utf-8")
170
+ } catch (error) {
171
+ if ((error as NodeJS.ErrnoException).code === "ENOENT") {
172
+ return null
173
+ }
174
+ throw error
175
+ }
176
+ try {
177
+ const parsed = JSON.parse(raw) as Partial<LiveUsageRecord>
178
+ if (typeof parsed.sessionId === "string" && parsed.status === "running") {
179
+ return parsed as LiveUsageRecord
180
+ }
181
+ return null
182
+ } catch {
183
+ // Loose validation: a partially-written snapshot is not a session.
184
+ return null
185
+ }
186
+ }
@@ -0,0 +1,161 @@
1
+ /**
2
+ * GitHub App authentication — installation access tokens.
3
+ *
4
+ * Wraps `@octokit/auth-app` (the GitHub-maintained library for exactly this;
5
+ * it handles the RS256 JWT signing and the POST /app/installations/{id}/
6
+ * access_tokens token-exchange protocol — hand-rolling that insecurely would
7
+ * be worse than taking the dependency). See docs/github-app-setup.md for the
8
+ * manual App registration the human owner must do first.
9
+ *
10
+ * Security model:
11
+ * - The durable secret is the App private key (from env, per
12
+ * docs/github-app-setup.md). Installation tokens are deliberately
13
+ * EPHEMERAL (1h validity) — we cache them in memory only, keyed by
14
+ * installation id, refresh on demand past expiry, and never persist a
15
+ * token to disk.
16
+ * - The default cache uses the library's own LRU (TTL = 59 min < GitHub's
17
+ * 1h token lifetime). Tests inject a fake cache + fake clock to prove the
18
+ * caching and refresh behavior without sleeping or hitting the network.
19
+ */
20
+
21
+ import { createAppAuth } from "@octokit/auth-app"
22
+ import { request as octokitRequest } from "@octokit/request"
23
+
24
+ /** Minimal shape of the library's installation-token auth result (the full
25
+ * `InstallationAccessTokenAuthentication` type is not re-exported from
26
+ * "@octokit/auth-app", so we declare the fields we consume). */
27
+ interface InstallationTokenResult {
28
+ token: string
29
+ expiresAt: string
30
+ }
31
+
32
+ /** Minimal callable shape of the auth strategy (matches AuthInterface's
33
+ * installation overload; the type itself isn't exported from the package). */
34
+ type InstallationAuth = (options: {
35
+ type: "installation"
36
+ installationId: number | string
37
+ refresh?: boolean
38
+ }) => Promise<InstallationTokenResult>
39
+
40
+ /** In-memory token cache interface (mirrors the shape the library's Cache
41
+ * option expects; ours is injectable + testable). */
42
+ export interface TokenCache {
43
+ get(key: string): string | undefined | Promise<string | undefined>
44
+ set(key: string, value: string): unknown | Promise<unknown>
45
+ }
46
+
47
+ interface Clock {
48
+ now(): number
49
+ }
50
+
51
+ export interface AppAuthConfig {
52
+ /** GitHub App numeric ID ($GITHUB_APP_ID). */
53
+ appId: string | number
54
+ /** App private key PEM text ($GITHUB_APP_PRIVATE_KEY). */
55
+ privateKey: string
56
+ /** Optional GitHub API base URL override (tests/mocks). */
57
+ baseUrl?: string
58
+ /** Injectable fetch for tests (default: global fetch). */
59
+ fetchImpl?: typeof fetch
60
+ /** Injectable clock for tests (default: Date.now). */
61
+ clock?: Clock
62
+ /** Optional injectable cache (default: the library's own in-memory LRU). */
63
+ cache?: TokenCache
64
+ }
65
+
66
+ export interface AppAuthClient {
67
+ /** A fresh (or cached, unexpired) installation access token. */
68
+ getInstallationToken(installationId: number | string): Promise<string>
69
+ }
70
+
71
+ /** 59 minutes — refresh a minute before GitHub's 1h token lifetime ends. */
72
+ export const TOKEN_TTL_MS = 59 * 60 * 1000
73
+ /** Cache-key prefix so mixed cache implementations stay namespaced. */
74
+ const CACHE_PREFIX = "github-app-installation-token:"
75
+
76
+ export class AppAuthError extends Error {
77
+ readonly installationId: number | string
78
+ constructor(message: string, installationId: number | string, options?: ErrorOptions) {
79
+ super(message, options)
80
+ this.name = "AppAuthError"
81
+ this.installationId = installationId
82
+ }
83
+ }
84
+
85
+ /**
86
+ * Build an AppAuthClient from App ID + private key (from env per
87
+ * docs/github-app-setup.md). The returned client caches installation tokens
88
+ * in memory, keyed by installation id, and refreshes them before expiry.
89
+ */
90
+ export function createAppAuthClient(config: AppAuthConfig): AppAuthClient {
91
+ const { appId, privateKey, baseUrl, fetchImpl, clock = { now: () => Date.now() }, cache } = config
92
+
93
+ // The library's Cache option expects { get, set } with string values; we
94
+ // wrap our own cache so the expiry bookkeeping stays in ONE place (here),
95
+ // which is what the tests assert against.
96
+ const libCache: Exclude<Parameters<typeof createAppAuth>[0]["cache"], undefined> = {
97
+ get: async (key: string) => (await cache?.get(key)) ?? "",
98
+ set: (key: string, value: string) => cache?.set(key, value) ?? undefined,
99
+ }
100
+
101
+ // The library's `request` option wants a full RequestInterface; build one
102
+ // from the package's own @octokit/request with our baseUrl/fetch injected
103
+ // (fetch is honored per-request by fetch-wrapper via options.request.fetch).
104
+ const request = octokitRequest.defaults({
105
+ ...(baseUrl ? { baseUrl } : {}),
106
+ ...(fetchImpl ? { request: { fetch: fetchImpl } } : {}),
107
+ })
108
+
109
+ const auth: InstallationAuth = createAppAuth({
110
+ appId,
111
+ privateKey,
112
+ request,
113
+ cache: libCache,
114
+ })
115
+
116
+ async function getInstallationToken(installationId: number | string): Promise<string> {
117
+ const key = `${CACHE_PREFIX}${installationId}`
118
+ const cached = await cache?.get(key)
119
+ if (typeof cached === "string") {
120
+ const parsed = JSON.parse(cached) as { token: string; expiresAt: number }
121
+ // Refresh BEFORE expiry (a 1-minute safety margin — see TOKEN_TTL_MS).
122
+ if (parsed.expiresAt > clock.now()) {
123
+ return parsed.token
124
+ }
125
+ }
126
+
127
+ let authentication
128
+ try {
129
+ authentication = await auth({
130
+ type: "installation",
131
+ installationId,
132
+ // `refresh: true` bypasses the library's OWN internal cache so
133
+ // OUR expiry bookkeeping (with the injectable clock) is what
134
+ // governs refreshes.
135
+ refresh: true,
136
+ })
137
+ } catch (err) {
138
+ throw new AppAuthError(
139
+ `failed to exchange App credentials for an installation access token (installation ${installationId}): ${
140
+ err instanceof Error ? err.message : String(err)
141
+ }`,
142
+ installationId,
143
+ { cause: err },
144
+ )
145
+ }
146
+
147
+ const token = authentication.token
148
+ if (!token) {
149
+ throw new AppAuthError(`installation token exchange returned no token (installation ${installationId})`, installationId)
150
+ }
151
+ const expiresAt = new Date(authentication.expiresAt).getTime()
152
+ if (!Number.isFinite(expiresAt)) {
153
+ throw new AppAuthError(`installation token exchange returned no expiry (installation ${installationId})`, installationId)
154
+ }
155
+ await cache?.set(key, JSON.stringify({ token, expiresAt }))
156
+
157
+ return token
158
+ }
159
+
160
+ return { getInstallationToken }
161
+ }