headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,221 @@
1
+ /**
2
+ * Phase 6 — per-session cost/time/iteration budget enforcement.
3
+ *
4
+ * `BudgetTracker` is the pure, unit-testable core of the per-session budget
5
+ * guardrail (spec 6.3): it sits inside a `HeadlessSession` (or any caller
6
+ * that makes LLM calls) and trips a `BudgetExceededError` when ANY of these
7
+ * limits is crossed:
8
+ *
9
+ * - `maxCostUsd` — accumulated estimated USD cost (tokens × pricing);
10
+ * - `maxDurationMs` — wall-clock elapsed since the tracker was created;
11
+ * - `maxIterations` — number of LLM calls (ticks) performed.
12
+ *
13
+ * Enforcement points (see `check` / `tick` / `record`):
14
+ * - `tick()` is called BEFORE each LLM call and re-checks elapsed time,
15
+ * iteration count AND accumulated cost (a previous call's `record()` may
16
+ * have pushed cost past the cap — the next `tick()` catches it before any
17
+ * further spend);
18
+ * - `record()` is called AFTER each LLM call with the provider's usage token
19
+ * counts and accumulates cost (checking the cost cap again).
20
+ *
21
+ * Pure by construction: no fs, no network, no global state. Time is
22
+ * injectable (`now`), so duration tests use a fake clock; pricing is
23
+ * injectable so cost tests don't depend on the env pricing file.
24
+ *
25
+ * When a budget is NOT configured the tracker is simply never created — zero
26
+ * behavior change (the session default is `budget: null`).
27
+ */
28
+
29
+ import { estimateCost, loadPricingTable, type PricingTable } from "./cost.js"
30
+
31
+ /** Per-session budget (all limits optional — only the set ones are enforced). */
32
+ export interface SessionBudget {
33
+ /** Max estimated USD spend for the session (accumulated across LLM calls). */
34
+ maxCostUsd?: number
35
+ /** Max wall-clock duration for the session, ms (checked before each call). */
36
+ maxDurationMs?: number
37
+ /** Max LLM-call iterations (a stricter cap than the loop's maxIterations). */
38
+ maxIterations?: number
39
+ }
40
+
41
+ export type BudgetLimitReason = "cost" | "duration" | "iterations"
42
+
43
+ /** Thrown by `tick()`/`record()` when a budget limit trips. */
44
+ export class BudgetExceededError extends Error {
45
+ readonly reason: BudgetLimitReason
46
+ readonly costUsd: number
47
+ readonly elapsedMs: number
48
+ readonly iterations: number
49
+
50
+ constructor(reason: BudgetLimitReason, costUsd: number, elapsedMs: number, iterations: number) {
51
+ super(`Budget exceeded: ${reason} (cost $${costUsd.toFixed(6)}, elapsed ${elapsedMs}ms, iterations ${iterations})`)
52
+ this.name = "BudgetExceededError"
53
+ this.reason = reason
54
+ this.costUsd = costUsd
55
+ this.elapsedMs = elapsedMs
56
+ this.iterations = iterations
57
+ }
58
+ }
59
+
60
+ export interface BudgetTrackerOptions {
61
+ /** Clock (default Date.now) — inject a fake for duration tests. */
62
+ now?: () => number
63
+ /** Pricing table (default: env-merged defaults) — inject for cost tests. */
64
+ pricing?: PricingTable
65
+ /**
66
+ * When false, record() never accumulates cost (stays 0 forever) and
67
+ * maxCostUsd checks never trip. Default true. Issue #144: a local-
68
+ * backend session has no real dollar cost — computing and logging a
69
+ * fabricated figure for it is noise at best, misleading at worst.
70
+ */
71
+ trackCost?: boolean
72
+ }
73
+
74
+ /** Non-throwing snapshot of the budget state (surfaced on SessionResult). */
75
+ export interface BudgetCheck {
76
+ ok: boolean
77
+ reason?: BudgetLimitReason
78
+ costUsd: number
79
+ elapsedMs: number
80
+ iterations: number
81
+ }
82
+
83
+ export class BudgetTracker {
84
+ private readonly budget: SessionBudget
85
+ private readonly now: () => number
86
+ private readonly pricing: PricingTable
87
+ private readonly trackCost: boolean
88
+ private readonly startedAt: number
89
+ private iterations = 0
90
+ private costUsd = 0
91
+ /** Cumulative ms spent paused (see pauseClock/resumeClock), excluded from elapsedMs. */
92
+ private blockedMs = 0
93
+ /** Set while paused (the `now()` value pauseClock() was called at); null when running. */
94
+ private blockStartedAt: number | null = null
95
+
96
+ constructor(budget: SessionBudget, options: BudgetTrackerOptions = {}) {
97
+ this.budget = budget
98
+ this.now = options.now ?? (() => Date.now())
99
+ this.pricing = options.pricing ?? loadPricingTable()
100
+ this.trackCost = options.trackCost ?? true
101
+ this.startedAt = this.now()
102
+ }
103
+
104
+ /**
105
+ * Decision escalation (workstream 2): call before blocking on an external
106
+ * answer (e.g. ask_followup_question waiting on `.harness.decision-answer`)
107
+ * so the wait doesn't count against `maxDurationMs`. Idempotent — a second
108
+ * call while already paused is a no-op.
109
+ */
110
+ pauseClock(): void {
111
+ if (this.blockStartedAt === null) {
112
+ this.blockStartedAt = this.now()
113
+ }
114
+ }
115
+
116
+ /**
117
+ * Resume the clock after a pause, folding the paused interval into
118
+ * `blockedMs`. Idempotent — a call while not paused is a no-op.
119
+ */
120
+ resumeClock(): void {
121
+ if (this.blockStartedAt !== null) {
122
+ this.blockedMs += Math.max(0, this.now() - this.blockStartedAt)
123
+ this.blockStartedAt = null
124
+ }
125
+ }
126
+
127
+ get elapsedMs(): number {
128
+ const raw = Math.max(0, this.now() - this.startedAt)
129
+ const inProgressBlock = this.blockStartedAt !== null ? Math.max(0, this.now() - this.blockStartedAt) : 0
130
+ return Math.max(0, raw - this.blockedMs - inProgressBlock)
131
+ }
132
+
133
+ get iterationCount(): number {
134
+ return this.iterations
135
+ }
136
+
137
+ get totalCostUsd(): number {
138
+ return this.costUsd
139
+ }
140
+
141
+ /**
142
+ * Call BEFORE each LLM call. Increments the iteration counter and re-checks
143
+ * duration, iterations and accumulated cost. Throws `BudgetExceededError`
144
+ * when a limit is crossed.
145
+ */
146
+ tick(): void {
147
+ this.iterations++
148
+ this.throwIfExceeded()
149
+ }
150
+
151
+ /**
152
+ * Call AFTER each LLM call with the provider's usage token counts.
153
+ * Accumulates the estimated cost and re-checks the cost cap. Throws
154
+ * `BudgetExceededError` when the cost limit trips.
155
+ */
156
+ record(input: { model: string; inputTokens?: number; outputTokens?: number; cachedTokens?: number }): void {
157
+ if (!this.trackCost) {
158
+ return
159
+ }
160
+ const cost = estimateCost({
161
+ model: input.model,
162
+ inputTokens: input.inputTokens ?? 0,
163
+ outputTokens: input.outputTokens ?? 0,
164
+ cachedTokens: input.cachedTokens ?? 0,
165
+ pricing: this.pricing,
166
+ })
167
+ this.costUsd += cost
168
+ this.throwIfExceeded()
169
+ }
170
+
171
+ /** Non-throwing snapshot: { ok, reason?, costUsd, elapsedMs, iterations }. */
172
+ check(): BudgetCheck {
173
+ const { reason } = this.exceededReason()
174
+ return {
175
+ ok: reason === undefined,
176
+ reason,
177
+ costUsd: this.costUsd,
178
+ elapsedMs: this.elapsedMs,
179
+ iterations: this.iterations,
180
+ }
181
+ }
182
+
183
+ private exceededReason(): { reason?: BudgetLimitReason } {
184
+ if (this.budget.maxDurationMs !== undefined && this.elapsedMs >= this.budget.maxDurationMs) {
185
+ return { reason: "duration" }
186
+ }
187
+ if (this.budget.maxIterations !== undefined && this.iterations > this.budget.maxIterations) {
188
+ return { reason: "iterations" }
189
+ }
190
+ if (this.budget.maxCostUsd !== undefined && this.costUsd >= this.budget.maxCostUsd) {
191
+ return { reason: "cost" }
192
+ }
193
+ return {}
194
+ }
195
+
196
+ private throwIfExceeded(): void {
197
+ const { reason } = this.exceededReason()
198
+ if (reason !== undefined) {
199
+ throw new BudgetExceededError(reason, this.costUsd, this.elapsedMs, this.iterations)
200
+ }
201
+ }
202
+ }
203
+
204
+ /**
205
+ * Build a `SessionBudget` from the harness env vars used by the CLI +
206
+ * run-worker.sh/run-qa.sh. Returns undefined when neither cost nor duration is
207
+ * set (budget OFF). Invalid values are ignored (validation lives in the CLI).
208
+ */
209
+ export function sessionBudgetFromEnv(env: NodeJS.ProcessEnv = process.env): SessionBudget | undefined {
210
+ const costRaw = env.HEADLESSCODE_MAX_COST_USD
211
+ const durationRaw = env.HEADLESSCODE_MAX_DURATION_MS
212
+ const cost = costRaw !== undefined && costRaw !== "" ? Number(costRaw) : undefined
213
+ const duration = durationRaw !== undefined && durationRaw !== "" ? Number(durationRaw) : undefined
214
+ if ((cost !== undefined && Number.isFinite(cost) && cost > 0) || (duration !== undefined && Number.isFinite(duration) && duration > 0)) {
215
+ return {
216
+ ...(cost !== undefined && Number.isFinite(cost) && cost > 0 ? { maxCostUsd: cost } : {}),
217
+ ...(duration !== undefined && Number.isFinite(duration) && duration > 0 ? { maxDurationMs: duration } : {}),
218
+ }
219
+ }
220
+ return undefined
221
+ }
@@ -0,0 +1,126 @@
1
+ /**
2
+ * Phase 6 — global concurrency guardrail (spec 6.3): a hard cap on how many
3
+ * harness sessions may run at once, enforced BEFORE anything is spawned.
4
+ *
5
+ * Two layers, both in this module:
6
+ *
7
+ * 1. `ConcurrencyLimiter` — an in-process counting semaphore. We deliberately
8
+ * chose FAIL-FAST over queueing for headless spawners: a queued spawner
9
+ * has nowhere to hold "waiting" work unless it is the watcher (which
10
+ * already has a durable pending mechanism), and a silently queued
11
+ * orchestrate run is worse than a loud abort. The watcher's "defer with
12
+ * durable pending state" is the queue — implemented in the watcher, not
13
+ * the limiter.
14
+ *
15
+ * 2. `activeSessionCount(...)` — a CROSS-PROCESS view of how many sessions
16
+ * are currently running, derived from the durable state files:
17
+ * - `.orchestrator-state.json` groups with status `spawned` | `running`;
18
+ * - `.watcher-state.json` entries with status `spawned` (in-flight
19
+ * write-ahead spawns).
20
+ * Each process (orchestrate, watcher, a cron watcher, a manual run) reads
21
+ * these files before spawning, so the cap holds even when several
22
+ * processes run concurrently against the same repo — the in-process
23
+ * limiter covers the window INSIDE one process (e.g. a sweep spawning
24
+ * several batches back-to-back), the state files cover the window ACROSS
25
+ * processes.
26
+ */
27
+
28
+ import * as path from "node:path"
29
+
30
+ import { loadStateSync, type OrchestratorState } from "../orchestrator/state.js"
31
+ import { loadWatcherStateSync, type WatcherState } from "../watcher/state.js"
32
+
33
+ /** Default global cap: HEADLESSCODE_MAX_CONCURRENT_SESSIONS or 3. */
34
+ export const DEFAULT_MAX_CONCURRENT_SESSIONS = 3
35
+
36
+ /**
37
+ * In-process counting semaphore with fail-fast acquisition. The cap counts
38
+ * SESSIONS (one worktree group = one harness session), not worktrees.
39
+ */
40
+ export class ConcurrencyLimiter {
41
+ readonly maxSessions: number
42
+ private active = 0
43
+
44
+ constructor(maxSessions: number) {
45
+ if (!Number.isInteger(maxSessions) || maxSessions <= 0) {
46
+ throw new Error(`ConcurrencyLimiter: maxSessions must be a positive integer (got ${maxSessions})`)
47
+ }
48
+ this.maxSessions = maxSessions
49
+ }
50
+
51
+ /**
52
+ * Try to reserve a slot. Fail-fast: returns { ok: false, reason } instead
53
+ * of queueing — a caller that cannot spawn right now must either defer
54
+ * (watcher → durable pending) or abort loudly (orchestrate), never wait
55
+ * silently. (Decision documented in docs/phase6-cloud.md.)
56
+ */
57
+ acquire(): { ok: boolean; reason?: string } {
58
+ if (this.active >= this.maxSessions) {
59
+ return {
60
+ ok: false,
61
+ reason: `concurrency cap reached: ${this.maxSessions} active session(s) (HEADLESSCODE_MAX_CONCURRENT_SESSIONS)`,
62
+ }
63
+ }
64
+ this.active++
65
+ return { ok: true }
66
+ }
67
+
68
+ /** Release a slot previously reserved with acquire(). Never goes below 0. */
69
+ release(): void {
70
+ this.active = Math.max(0, this.active - 1)
71
+ }
72
+
73
+ /** Current number of reserved (active) slots in this process. */
74
+ current(): number {
75
+ return this.active
76
+ }
77
+ }
78
+
79
+ /** Parsed env cap: HEADLESSCODE_MAX_CONCURRENT_SESSIONS (default 3). */
80
+ export function maxConcurrentSessionsFromEnv(env: NodeJS.ProcessEnv = process.env): number {
81
+ const raw = env.HEADLESSCODE_MAX_CONCURRENT_SESSIONS
82
+ if (raw === undefined || raw === "") {
83
+ return DEFAULT_MAX_CONCURRENT_SESSIONS
84
+ }
85
+ const n = Number(raw)
86
+ return Number.isInteger(n) && n > 0 ? n : DEFAULT_MAX_CONCURRENT_SESSIONS
87
+ }
88
+
89
+ /**
90
+ * Cross-process active-session count: orchestrator groups in `spawned`/
91
+ * `running` + watcher entries in `spawned` (write-ahead, spawn in flight).
92
+ * Pass either state as null/undefined when that pipeline isn't in use.
93
+ */
94
+ export function activeSessionCount(
95
+ orchState?: OrchestratorState | null,
96
+ watcherState?: WatcherState | null,
97
+ ): number {
98
+ let count = 0
99
+ if (orchState) {
100
+ for (const group of orchState.groups) {
101
+ if (group.status === "spawned" || group.status === "running") {
102
+ count++
103
+ }
104
+ }
105
+ }
106
+ if (watcherState) {
107
+ for (const entry of Object.values(watcherState.processed)) {
108
+ if (entry.status === "spawned") {
109
+ count++
110
+ }
111
+ }
112
+ }
113
+ return count
114
+ }
115
+
116
+ /**
117
+ * Convenience: load BOTH durable state files for a repo root and return the
118
+ * combined active count. Missing files are tolerated (fresh default state).
119
+ * Paths follow the repo conventions: `<repoRoot>/.worktrees/`.
120
+ */
121
+ export function activeSessionCountForRepo(repoRoot: string): number {
122
+ const worktrees = path.join(path.resolve(repoRoot), ".worktrees")
123
+ const orchState = loadStateSync(path.join(worktrees, ".orchestrator-state.json"))
124
+ const watcherState = loadWatcherStateSync(path.join(worktrees, ".watcher-state.json"))
125
+ return activeSessionCount(orchState, watcherState)
126
+ }
@@ -0,0 +1,309 @@
1
+ /**
2
+ * Phase 6 — model pricing + LLM cost estimation (the numeric heart of the
3
+ * per-session cost budget).
4
+ *
5
+ * The harness pays OpenRouter per token (input + output). To enforce a
6
+ * per-session cost cap without post-hoc reconciliation we estimate the USD
7
+ * cost of every LLM call from the `usage` object the provider returns
8
+ * (prompt/completion tokens) × the model's price.
9
+ *
10
+ * Pricing table: USD per 1M tokens, keyed by model id. Defaults cover the
11
+ * models this project actually uses (DeepSeek family via OpenRouter) plus one
12
+ * general model as a sanity reference. The table is overridable:
13
+ *
14
+ * - in-code: pass a `PricingTable` to `estimateCost` / `accumulateCost` /
15
+ * `BudgetTracker` (the constructor option);
16
+ * - via env: `HEADLESSCODE_PRICING_JSON` = path to a JSON file of the same
17
+ * shape (`{ "<model>": { "input": <per1M USD>, "output": <per1M USD> } }`),
18
+ * deep-merged over the defaults (per-model override; unlisted models keep
19
+ * the built-in price).
20
+ *
21
+ * Unknown models fall back to a CONSERVATIVE general rate (`FALLBACK_MODEL_PRICE`)
22
+ * rather than $0 — a runaway session on a model we don't have a price for must
23
+ * still be capped, not silently free.
24
+ *
25
+ * Prices below are the list rates as of the 2026-08 baseline; they are input
26
+ * to a GUARDRAIL (cap enforcement), so being slightly stale is safe (the cap
27
+ * direction — "did we cross the budget" — is what matters).
28
+ */
29
+
30
+ import * as fs from "node:fs"
31
+ import * as path from "node:path"
32
+
33
+ /** USD per 1M tokens for one model. */
34
+ export interface ModelPrice {
35
+ /** Price per 1M INPUT (prompt) tokens, USD. */
36
+ input: number
37
+ /** Price per 1M OUTPUT (completion) tokens, USD. */
38
+ output: number
39
+ /**
40
+ * Price per 1M CACHED input tokens, USD (a provider-side prompt-cache hit —
41
+ * see LlmResponse.usage.cachedTokens). Optional and deliberately NOT
42
+ * defaulted to a discounted guess: when absent, cached tokens are priced
43
+ * at the full `input` rate, i.e. zero behavior change from before cache
44
+ * accounting existed. Only set this once you've confirmed the model's
45
+ * real cache-hit rate (e.g. from OpenRouter/provider docs) — this is a
46
+ * cost GUARDRAIL, so a wrong optimistic discount could let real spend
47
+ * exceed a configured cap.
48
+ */
49
+ cacheRead?: number
50
+ }
51
+
52
+ /** Model id → price. */
53
+ export type PricingTable = Record<string, ModelPrice>
54
+
55
+ /**
56
+ * Default pricing table (OpenRouter list rates, USD per 1M tokens).
57
+ *
58
+ * - `deepseek/deepseek-v4-flash-0731` — the harness default (Phase 1/2/5
59
+ * workers; see `DEFAULT_MODEL` in src/llm/openrouter.ts). Priced the same
60
+ * as the prior default, `deepseek/deepseek-v4-flash` (kept below for
61
+ * override compat) — same model line, dated point-release id. The rate is
62
+ * the OFFICIAL DeepSeek provider's own rate specifically (verified
63
+ * against OpenRouter's own
64
+ * `/api/v1/models/deepseek/deepseek-v4-flash/endpoints` on 2026-08-01,
65
+ * cross-checked against OpenRouter's own pricing UI). This price is only
66
+ * actually correct because `src/llm/openrouter.ts` pins routing to this
67
+ * exact provider (`provider: { order: ["deepseek"], allow_fallbacks:
68
+ * false }`) for deepseek/* models — OTHER routed endpoints behind the
69
+ * same model id (DeepInfra, Baidu, Mancer, ...) have meaningfully
70
+ * different prices, especially for cache reads, so this number would be
71
+ * wrong/unverifiable without that pin. If the pin is ever removed, this
72
+ * price must be revisited, not left as a stale guess.
73
+ * Missing entirely from this table before was the actual root cause of a
74
+ * real incident: sessions silently fell through to `FALLBACK_MODEL_PRICE`
75
+ * (7x+ this model's real input rate), producing wildly inflated internal
76
+ * cost estimates and at least one false-positive budget abort. A second,
77
+ * smaller error (`cacheRead` off by 10x — $0.028 instead of the real
78
+ * $0.0028/1M) was caught and fixed the same day.
79
+ * - `deepseek/deepseek-v4-flash` — the previous default, kept priced so an
80
+ * explicit override to the un-dated id still estimates correctly.
81
+ * - `deepseek/deepseek-chat` — the legacy default, still priced so an
82
+ * explicit `OPENROUTER_MODEL=deepseek/deepseek-chat` override accounts
83
+ * correctly.
84
+ * - `deepseek/deepseek-reasoner` — the reasoning variant used for review/QA.
85
+ * - `anthropic/claude-3.5-sonnet` — a general-purpose reference model.
86
+ * - `qwen/qwen3-embedding-4b` / `qwen/qwen3-embedding-8b` — the codebase-index
87
+ * embedding models (src/codesearch/embedder.ts; 8b is the default since
88
+ * the central-store round, 4b kept priced so a legacy index/override still
89
+ * estimates correctly). Embedding responses carry no completion tokens
90
+ * (output is always 0). The rates are CONSERVATIVE input-only estimates:
91
+ * $0.15/1M for 4b was observed live on OpenRouter's Google embedding
92
+ * endpoint (2026-08-01), and 8b is estimated at $0.30/1M (2x the 4b rate,
93
+ * parameter-scaled) because embedding models are absent from OpenRouter's
94
+ * /api/v1/models listing so neither list price is API-verifiable (see
95
+ * src/codesearch/embedder.ts's header). Pricing a guardrail means
96
+ * fail-closed: estimate rather than $0, so an index build on a large repo
97
+ * still counts against a cost cap.
98
+ */
99
+ export const DEFAULT_PRICING_TABLE: PricingTable = {
100
+ "deepseek/deepseek-chat": { input: 0.27, output: 1.1 },
101
+ "deepseek/deepseek-reasoner": { input: 0.55, output: 2.19 },
102
+ "deepseek/deepseek-v4-flash-0731": { input: 0.14, output: 0.28, cacheRead: 0.0028 },
103
+ "deepseek/deepseek-v4-flash": { input: 0.14, output: 0.28, cacheRead: 0.0028 },
104
+ "anthropic/claude-3.5-sonnet": { input: 3.0, output: 15.0 },
105
+ "qwen/qwen3-embedding-4b": { input: 0.15, output: 0.15 },
106
+ "qwen/qwen3-embedding-8b": { input: 0.3, output: 0.3 },
107
+ // Cloud vision captioning (src/vision/describe.ts). Live OpenRouter list
108
+ // rates pulled 2026-08-02 for image-input models; the 12b is the default
109
+ // (cheapest candidate that wasn't materially worse than the quality anchor
110
+ // in the image-support evaluation — see plans/image-support.md), and the
111
+ // 27b is priced too so an override to the quality anchor still estimates
112
+ // accurately instead of falling through to FALLBACK_MODEL_PRICE. Both were
113
+ // verified live against the API during the evaluation.
114
+ "google/gemma-3-12b-it": { input: 0.05, output: 0.15 },
115
+ "google/gemma-3-27b-it": { input: 0.08, output: 0.45 },
116
+ }
117
+
118
+ /**
119
+ * Conservative fallback for models missing from the table (USD per 1M).
120
+ * Deliberately a mid-range general rate, NOT $0 — unknown models must still
121
+ * count against the budget (fail-closed: over-estimate slightly rather than
122
+ * underestimate, so a cost cap is never bypassed by an unlisted model id).
123
+ */
124
+ export const FALLBACK_MODEL_PRICE: ModelPrice = { input: 2.0, output: 8.0 }
125
+
126
+ /** One endpoint entry of OpenRouter's `/api/v1/models/<id>/endpoints` response. */
127
+ export interface EndpointPricingEntry {
128
+ /** Provider slug as OpenRouter reports it (e.g. "DeepSeek", "DeepInfra"). */
129
+ provider_name?: unknown
130
+ /** Per-token prices as STRINGS, USD per token (verified live 2026-08-04). */
131
+ pricing?: {
132
+ prompt?: unknown
133
+ completion?: unknown
134
+ input_cache_read?: unknown
135
+ }
136
+ }
137
+
138
+ /** Parse a per-token USD string like "0.00000014" into per-1M-token USD. */
139
+ function perTokenToPerMillion(value: unknown): number | undefined {
140
+ if (typeof value !== "string" && typeof value !== "number") {
141
+ return undefined
142
+ }
143
+ const n = typeof value === "string" ? Number(value) : value
144
+ if (!Number.isFinite(n) || n < 0) {
145
+ return undefined
146
+ }
147
+ return n * 1_000_000
148
+ }
149
+
150
+ /**
151
+ * Parse live endpoint pricing into a `ModelPrice` (USD per 1M tokens).
152
+ *
153
+ * Selection rules, mirroring how the request actually routes:
154
+ * - When `providerPreference` is given (e.g. "DeepSeek" for the deepseek/*
155
+ * chat pin), the FIRST endpoint whose provider_name matches (case-
156
+ * insensitive) is used verbatim — that is the price actually charged.
157
+ * - Otherwise the MAX of each field across all endpoints that expose it is
158
+ * used. This is a deliberate fail-closed choice: for auto-routed models the
159
+ * harness cannot know which endpoint served a call, and this table is a
160
+ * GUARDRAIL (the codebase's FALLBACK_MODEL_PRICE rationale) — over-
161
+ * estimating is safe, under-estimating lets a cap be bypassed.
162
+ *
163
+ * Returns undefined when no endpoint exposes usable pricing.
164
+ */
165
+ export function parseEndpointPricing(
166
+ endpoints: EndpointPricingEntry[],
167
+ providerPreference?: string,
168
+ ): ModelPrice | undefined {
169
+ let price: ModelPrice | undefined
170
+ for (const ep of endpoints) {
171
+ const pricing = ep?.pricing
172
+ if (!pricing || typeof pricing !== "object") {
173
+ continue
174
+ }
175
+ const parsed: ModelPrice = {
176
+ input: perTokenToPerMillion(pricing.prompt) ?? 0,
177
+ output: perTokenToPerMillion(pricing.completion) ?? 0,
178
+ }
179
+ if (parsed.input === 0 && parsed.output === 0) {
180
+ continue
181
+ }
182
+ const cacheRead = perTokenToPerMillion(pricing.input_cache_read)
183
+ if (cacheRead !== undefined) {
184
+ parsed.cacheRead = cacheRead
185
+ }
186
+ // Provider-preference match wins outright (first match — a match
187
+ // returns immediately, so reaching the end without returning means no
188
+ // usable match existed).
189
+ if (providerPreference) {
190
+ const name = typeof ep?.provider_name === "string" ? ep.provider_name : ""
191
+ if (name.toLowerCase() === providerPreference.toLowerCase()) {
192
+ return parsed
193
+ }
194
+ }
195
+ // Otherwise keep the conservative per-field max.
196
+ price = {
197
+ input: Math.max(price?.input ?? 0, parsed.input),
198
+ output: Math.max(price?.output ?? 0, parsed.output),
199
+ ...(cacheRead !== undefined
200
+ ? { cacheRead: Math.max(price?.cacheRead ?? 0, cacheRead) }
201
+ : price?.cacheRead !== undefined
202
+ ? { cacheRead: price.cacheRead }
203
+ : {}),
204
+ }
205
+ }
206
+ // A preference was requested but no endpoint matched it — return nothing
207
+ // so the caller falls back to the hardcoded table (for deepseek/* that
208
+ // table IS the official endpoint's rate; using some other host's numbers
209
+ // for a pinned model would be wrong).
210
+ if (providerPreference) {
211
+ return undefined
212
+ }
213
+ return price
214
+ }
215
+
216
+ /** Merge a live-resolved price into a pricing table under a model id. */
217
+ export function mergeLivePrice(table: PricingTable, model: string, price: ModelPrice): PricingTable {
218
+ return { ...table, [model]: price }
219
+ }
220
+
221
+ /**
222
+ * Load the effective pricing table: defaults deep-merged with overrides from
223
+ * `HEADLESSCODE_PRICING_JSON` (when set). A missing/unreadable/malformed file
224
+ * throws — a broken pricing override must fail loudly, never silently weaken
225
+ * a cost cap.
226
+ */
227
+ export function loadPricingTable(
228
+ env: NodeJS.ProcessEnv = process.env,
229
+ base: PricingTable = DEFAULT_PRICING_TABLE,
230
+ ): PricingTable {
231
+ const pricingPath = env.HEADLESSCODE_PRICING_JSON
232
+ if (!pricingPath) {
233
+ return base
234
+ }
235
+ const file = path.resolve(pricingPath)
236
+ let raw: string
237
+ try {
238
+ raw = fs.readFileSync(file, "utf-8")
239
+ } catch (err) {
240
+ throw new Error(
241
+ `HEADLESSCODE_PRICING_JSON: cannot read pricing file '${file}': ${err instanceof Error ? err.message : String(err)}`,
242
+ )
243
+ }
244
+ let parsed: unknown
245
+ try {
246
+ parsed = JSON.parse(raw)
247
+ } catch (err) {
248
+ throw new Error(
249
+ `HEADLESSCODE_PRICING_JSON: invalid JSON in '${file}': ${err instanceof Error ? err.message : String(err)}`,
250
+ )
251
+ }
252
+ if (parsed === null || typeof parsed !== "object" || Array.isArray(parsed)) {
253
+ throw new Error(`HEADLESSCODE_PRICING_JSON: '${file}' must be a JSON object of {model: {input, output}}`)
254
+ }
255
+ const overrides: PricingTable = {}
256
+ for (const [model, value] of Object.entries(parsed)) {
257
+ const v = value as { input?: unknown; output?: unknown }
258
+ if (typeof v.input === "number" && Number.isFinite(v.input) && v.input >= 0 && typeof v.output === "number" && Number.isFinite(v.output) && v.output >= 0) {
259
+ overrides[model] = { input: v.input, output: v.output }
260
+ }
261
+ }
262
+ return { ...base, ...overrides }
263
+ }
264
+
265
+ /** Resolve the price for a model (fallback when unlisted). */
266
+ export function priceFor(model: string, pricing: PricingTable = DEFAULT_PRICING_TABLE): ModelPrice {
267
+ return pricing[model] ?? FALLBACK_MODEL_PRICE
268
+ }
269
+
270
+ export interface CostInput {
271
+ model: string
272
+ inputTokens?: number
273
+ outputTokens?: number
274
+ /**
275
+ * Prompt tokens served from the provider's cache (a SUBSET of
276
+ * inputTokens, not additional — see LlmResponse.usage.cachedTokens).
277
+ * Priced at price.cacheRead when set, else at the full input rate.
278
+ */
279
+ cachedTokens?: number
280
+ /** Explicit pricing table (default: env-merged defaults). */
281
+ pricing?: PricingTable
282
+ }
283
+
284
+ /**
285
+ * Estimate the USD cost of one LLM call from its usage token counts.
286
+ * cost = (inputTokens - cachedTokens) × price.input / 1e6
287
+ * + cachedTokens × (price.cacheRead ?? price.input) / 1e6
288
+ * + outputTokens × price.output / 1e6
289
+ * `cachedTokens` is clamped to `inputTokens` (a provider reporting more
290
+ * cached than total prompt tokens would otherwise produce a negative
291
+ * "uncached" count).
292
+ */
293
+ export function estimateCost(input: CostInput): number {
294
+ const { model, inputTokens = 0, outputTokens = 0 } = input
295
+ const table = input.pricing ?? loadPricingTable()
296
+ const price = priceFor(model, table)
297
+ const cachedTokens = Math.min(Math.max(input.cachedTokens ?? 0, 0), inputTokens)
298
+ const uncachedTokens = inputTokens - cachedTokens
299
+ const cacheRate = price.cacheRead ?? price.input
300
+ return (uncachedTokens * price.input + cachedTokens * cacheRate + outputTokens * price.output) / 1_000_000
301
+ }
302
+
303
+ /** Sum the estimated cost across multiple calls (runs) — the session total. */
304
+ export function accumulateCost(
305
+ runs: Array<{ model: string; inputTokens?: number; outputTokens?: number; cachedTokens?: number }>,
306
+ pricing?: PricingTable,
307
+ ): number {
308
+ return runs.reduce((sum, run) => sum + estimateCost({ ...run, pricing }), 0)
309
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * Phase 6 — guardrails (spec 6.3): per-session cost/time/iteration budgets and
3
+ * the global concurrent-session cap. See docs/phase6-cloud.md.
4
+ */
5
+
6
+ export * from "./cost.js"
7
+ export * from "./budget.js"
8
+ export * from "./concurrency.js"