headlesscode 1.0.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/ATTRIBUTION.md +53 -0
  2. package/CODE_OF_CONDUCT.md +130 -0
  3. package/CONTRIBUTING.md +107 -0
  4. package/LICENSE +202 -0
  5. package/README.md +486 -0
  6. package/SECURITY.md +211 -0
  7. package/bin/headlesscode.mjs +83 -0
  8. package/package.json +63 -0
  9. package/shared/prompts/review-mode-prompt-short.md +93 -0
  10. package/shared/prompts/review-mode-prompt.md +281 -0
  11. package/shared/rules-code/rules.md +22 -0
  12. package/shared/stacks/cpp/rules.md +30 -0
  13. package/shared/stacks/fastapi/rules.md +30 -0
  14. package/shared/stacks/javascript/rules.md +37 -0
  15. package/shared/stacks/postgresql/rules.md +31 -0
  16. package/shared/stacks/python/rules.md +35 -0
  17. package/shared/stacks/react/rules.md +11 -0
  18. package/shared/stacks/typescript/rules.md +10 -0
  19. package/src/budget/budget.ts +221 -0
  20. package/src/budget/concurrency.ts +126 -0
  21. package/src/budget/cost.ts +309 -0
  22. package/src/budget/index.ts +8 -0
  23. package/src/checkpoints/cli.ts +256 -0
  24. package/src/checkpoints/service.ts +227 -0
  25. package/src/cli.ts +1535 -0
  26. package/src/cloud/docker-provider.ts +334 -0
  27. package/src/cloud/provider.ts +300 -0
  28. package/src/codeintel/call-graph.ts +78 -0
  29. package/src/codeintel/find-references.ts +123 -0
  30. package/src/codeintel/go-to-definition.ts +193 -0
  31. package/src/codeintel/handlers.ts +190 -0
  32. package/src/codeintel/import-graph.ts +173 -0
  33. package/src/codeintel/outline.ts +180 -0
  34. package/src/codeintel/position.ts +77 -0
  35. package/src/codeintel/program.ts +350 -0
  36. package/src/codeintel/rename-symbol.ts +213 -0
  37. package/src/codeintel/tools.ts +280 -0
  38. package/src/codemap/build.ts +135 -0
  39. package/src/codemap/cli.ts +190 -0
  40. package/src/codemap/extract.ts +339 -0
  41. package/src/codemap/files.ts +236 -0
  42. package/src/codemap/fingerprint.ts +65 -0
  43. package/src/codemap/flows.ts +62 -0
  44. package/src/codemap/html.ts +451 -0
  45. package/src/codemap/lock.ts +80 -0
  46. package/src/codemap/types.ts +101 -0
  47. package/src/codesearch/airunner-embedder.ts +185 -0
  48. package/src/codesearch/chunk.ts +339 -0
  49. package/src/codesearch/cli.ts +223 -0
  50. package/src/codesearch/embedder.ts +332 -0
  51. package/src/codesearch/files.ts +280 -0
  52. package/src/codesearch/index.ts +469 -0
  53. package/src/codesearch/ollama-embedder.ts +205 -0
  54. package/src/codesearch/search.ts +141 -0
  55. package/src/codesearch/types.ts +100 -0
  56. package/src/config/mode-models.ts +218 -0
  57. package/src/dashboard/aggregate.ts +364 -0
  58. package/src/dashboard/chat-thread.ts +141 -0
  59. package/src/dashboard/checkpoints.ts +124 -0
  60. package/src/dashboard/cli.ts +193 -0
  61. package/src/dashboard/codemap.ts +44 -0
  62. package/src/dashboard/files.ts +121 -0
  63. package/src/dashboard/page.ts +2803 -0
  64. package/src/dashboard/self-improvement-metrics.ts +282 -0
  65. package/src/dashboard/server.ts +1103 -0
  66. package/src/dashboard/session-launch.ts +310 -0
  67. package/src/dashboard/timeline.ts +273 -0
  68. package/src/dashboard/tool-exec.ts +107 -0
  69. package/src/dashboard/trend-cli.ts +141 -0
  70. package/src/dashboard/trend.ts +413 -0
  71. package/src/decision-proxy/cli.ts +261 -0
  72. package/src/decision-proxy/proxy.ts +569 -0
  73. package/src/deploy/gate-cli.ts +147 -0
  74. package/src/deploy/gate.ts +254 -0
  75. package/src/engine/condense.ts +512 -0
  76. package/src/engine/events.ts +428 -0
  77. package/src/engine/handoff.ts +71 -0
  78. package/src/engine/lazy-tools.ts +160 -0
  79. package/src/engine/local-explore.ts +653 -0
  80. package/src/engine/logger.ts +96 -0
  81. package/src/engine/loop.ts +5517 -0
  82. package/src/engine/parser.ts +347 -0
  83. package/src/engine/prompt.ts +860 -0
  84. package/src/engine/reports.ts +47 -0
  85. package/src/engine/stacks.ts +448 -0
  86. package/src/engine/types.ts +291 -0
  87. package/src/engine/usage.ts +186 -0
  88. package/src/github/app-auth.ts +161 -0
  89. package/src/github/cli.ts +448 -0
  90. package/src/github/installations.ts +133 -0
  91. package/src/github/pr.ts +321 -0
  92. package/src/github/provision.ts +118 -0
  93. package/src/github/push.ts +122 -0
  94. package/src/index-util.ts +50 -0
  95. package/src/index.ts +81 -0
  96. package/src/init/cli.ts +248 -0
  97. package/src/init/gitignore.ts +74 -0
  98. package/src/llm/ollama.ts +308 -0
  99. package/src/llm/openrouter.ts +868 -0
  100. package/src/llm/preflight.ts +367 -0
  101. package/src/llm/transcript-capture.ts +84 -0
  102. package/src/memory/embed.ts +110 -0
  103. package/src/memory/index.ts +22 -0
  104. package/src/memory/local.ts +259 -0
  105. package/src/memory/summarizer.ts +283 -0
  106. package/src/memory/types.ts +153 -0
  107. package/src/memory/uwuchat.ts +157 -0
  108. package/src/migrate/cli.ts +115 -0
  109. package/src/orchestrator/analyze-cli.ts +104 -0
  110. package/src/orchestrator/auto-split.ts +206 -0
  111. package/src/orchestrator/cleanup.ts +1003 -0
  112. package/src/orchestrator/cli.ts +3571 -0
  113. package/src/orchestrator/cost-estimate.ts +564 -0
  114. package/src/orchestrator/cost-history-cli.ts +242 -0
  115. package/src/orchestrator/cost-history.ts +397 -0
  116. package/src/orchestrator/git-sync.ts +250 -0
  117. package/src/orchestrator/index.ts +153 -0
  118. package/src/orchestrator/log-analysis.ts +0 -0
  119. package/src/orchestrator/merge-check.ts +108 -0
  120. package/src/orchestrator/pipeline.ts +411 -0
  121. package/src/orchestrator/resume.ts +1940 -0
  122. package/src/orchestrator/reviewer.ts +503 -0
  123. package/src/orchestrator/split.ts +296 -0
  124. package/src/orchestrator/state.ts +542 -0
  125. package/src/orchestrator/status.ts +697 -0
  126. package/src/orchestrator/verification-gate.ts +134 -0
  127. package/src/orchestrator/watch.ts +898 -0
  128. package/src/permissions/commands.ts +1083 -0
  129. package/src/permissions/config.ts +241 -0
  130. package/src/permissions/index.ts +12 -0
  131. package/src/permissions/protected-files.ts +96 -0
  132. package/src/permissions/store-protection.ts +272 -0
  133. package/src/project-store.ts +648 -0
  134. package/src/projects/cli.ts +382 -0
  135. package/src/qa/qa.ts +487 -0
  136. package/src/tools/browser/handler.ts +346 -0
  137. package/src/tools/browser/service.ts +406 -0
  138. package/src/tools/browser/smoke.ts +78 -0
  139. package/src/tools/browser/tool.ts +99 -0
  140. package/src/tools/executor.ts +2575 -0
  141. package/src/tools/language-detect.ts +183 -0
  142. package/src/tools/output-summarizer.ts +369 -0
  143. package/src/tools/run-tests.ts +302 -0
  144. package/src/tools/set-indentation-tool.ts +49 -0
  145. package/src/tools/test-selection.ts +160 -0
  146. package/src/vendor/tests/smoke.ts +103 -0
  147. package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
  148. package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
  149. package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
  150. package/src/vendor/zoo-code/shim/os-name.ts +18 -0
  151. package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
  152. package/src/vendor/zoo-code/shim/vscode.ts +76 -0
  153. package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
  154. package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
  155. package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
  156. package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
  157. package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
  158. package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
  159. package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
  160. package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
  161. package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
  162. package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
  163. package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
  164. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
  165. package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
  166. package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
  167. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
  168. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
  169. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
  170. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
  171. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
  172. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
  173. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
  174. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
  175. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
  176. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
  177. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
  178. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
  179. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
  180. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
  181. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
  182. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
  183. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
  184. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
  185. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
  186. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
  187. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
  188. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
  189. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
  190. package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
  191. package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
  192. package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
  193. package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
  194. package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
  195. package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
  196. package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
  197. package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
  198. package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
  199. package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
  200. package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
  201. package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
  202. package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
  203. package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
  204. package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
  205. package/src/vendor/zoo-code/src/shared/language.ts +43 -0
  206. package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
  207. package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
  208. package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
  209. package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
  210. package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
  211. package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
  212. package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
  213. package/src/vendor/zoo-code/src/utils/object.ts +18 -0
  214. package/src/vendor/zoo-code/src/utils/path.ts +94 -0
  215. package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
  216. package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
  217. package/src/vendor/zoo-code/types/global-settings.ts +19 -0
  218. package/src/vendor/zoo-code/types/index.ts +22 -0
  219. package/src/vendor/zoo-code/types/message.ts +375 -0
  220. package/src/vendor/zoo-code/types/mode.ts +241 -0
  221. package/src/vendor/zoo-code/types/todo.ts +19 -0
  222. package/src/vendor/zoo-code/types/tool-params.ts +116 -0
  223. package/src/vendor/zoo-code/types/tool.ts +67 -0
  224. package/src/vendor/zoo-code/types/vscode.ts +84 -0
  225. package/src/vision/describe.ts +242 -0
  226. package/src/vision/tool.ts +91 -0
  227. package/src/watcher/cli.ts +369 -0
  228. package/src/watcher/github.ts +304 -0
  229. package/src/watcher/index.ts +59 -0
  230. package/src/watcher/state.ts +254 -0
  231. package/src/watcher/watch.ts +562 -0
  232. package/tsconfig.json +18 -0
@@ -0,0 +1,564 @@
1
+ /**
2
+ * Cost estimation from recorded cost history (issue #16) — the "future half"
3
+ * of cost-history.ts: once enough groups have reached a terminal status and
4
+ * been recorded, use those records to estimate the likely cost of a NEW task
5
+ * before spawning it.
6
+ *
7
+ * Matching signal is the issue SHAPE (split.ts's issueShape: hot / split /
8
+ * coverage / test / docs / refactor / generic), which split.ts already uses
9
+ * to batch same-shape mechanical work — an issue whose shape matches past
10
+ * ones inherits their recorded cost profile. A STRONGER signal is an exact
11
+ * issue-number match (the same task was worked on in a previous round) and
12
+ * always wins when present.
13
+ *
14
+ * Estimation is deliberately rough and honest about data scarcity: a shape
15
+ * needs at least MIN_SHAPE_SAMPLES recorded groups before a numeric estimate
16
+ * is offered (below that, the dry-run prints "insufficient data (N
17
+ * sample(s))"), because a 1-2 sample median is not an estimate, it's an
18
+ * anecdote — the issue itself says to revisit "once a meaningful number of
19
+ * records exist". Historical records are per-GROUP totals that can span
20
+ * several issues, so costs/iterations are normalized per-issue
21
+ * (total / max(issues.length, 1)) before matching against a NEW issue.
22
+ *
23
+ * Verification-intensity multiplier (issue #124): test-shaped issues (and
24
+ * issues whose remediation mentions CI/GitHub Actions) are verification-
25
+ * heavy — a diff that adds CI config forces real GitHub-runner round-trips,
26
+ * which dominated round-2026-08-17's cost. The thin samples for those shapes
27
+ * under-predicted ~2.6x (preflight ~$0.79 vs ~$2.09 actual; w13's test/hot
28
+ * group estimated ~$0.05-$0.27, actual $1.03). A shape-match estimate's
29
+ * COST is therefore scaled up until real samples accumulate — the recorded
30
+ * median of a verification-heavy shape is not trusted as-is while its
31
+ * samples are thin (see TEST_SHAPE_VERIFICATION_MULTIPLIER).
32
+ *
33
+ * Flagged CI/config higher bound (issue #125): complementary to the
34
+ * verification multiplier above, but a SEPARATE mechanism — an issue whose
35
+ * title/body itself names CI/config work (CI_CONFIG_RE) gets an additional
36
+ * flagged higher-bound estimate (its per-issue median times
37
+ * CI_CONFIG_HIGHER_BOUND_MULTIPLIER, rolled into the group's higherBoundUsd)
38
+ * surfaced alongside — not instead of — the primary cost estimate, since
39
+ * CI/test-infrastructure work historically costs 2-3x its shape median and
40
+ * the plain shape estimate under-predicted rounds containing it
41
+ * (round-2026-08-17: ~$0.79 estimated vs ~$2.09 actual).
42
+ *
43
+ * Issue #126: every group estimate carries a lowConfidence flag, set when
44
+ * the estimate is built on thin or missing history (unmatched issues whose
45
+ * cost is excluded from the total, an under-sampled direct match, or no
46
+ * usable history at all) — buildEstimateSection renders it as a prominent
47
+ * LOW-CONFIDENCE marker so preflight output can't present an unreliable
48
+ * total as if it were solid.
49
+ *
50
+ * Pure functions (no fs, no network) — fully unit-testable; the only I/O is
51
+ * readCostHistory() at the call site (cli.ts's dry-run).
52
+ */
53
+
54
+ import type { CostHistoryRecord } from "./cost-history.js"
55
+ import { issueShape, type SplitIssue, type WorktreeSpec } from "./split.js"
56
+
57
+ /**
58
+ * Minimum recorded groups of a shape before a numeric cost/iteration
59
+ * estimate is offered. Below this the estimate is suppressed in favor of an
60
+ * explicit "insufficient data" note. Exact issue-number matches (direct
61
+ * history of the SAME task) are exempt — one record of the very task being
62
+ * estimated is meaningful on its own (though it counts as LOW-CONFIDENCE,
63
+ * issue #126 — see estimateGroup).
64
+ */
65
+ export const MIN_SHAPE_SAMPLES = 3
66
+
67
+ /**
68
+ * Issue #124: multiplier applied to a verification-heavy issue's shape-match
69
+ * COST estimate while that shape's recorded samples are thin (the observed
70
+ * under-prediction was ~2.6x in round-2026-08-17, dominated by test/CI
71
+ * shapes). Maintainers' discretion per the issue; 2.5 sits between the 2-3x
72
+ * observed misses. Tapers toward 1 as real samples accumulate — the median
73
+ * of a well-sampled verification-heavy shape already reflects real
74
+ * verification cost.
75
+ */
76
+ export const TEST_SHAPE_VERIFICATION_MULTIPLIER = 2.5
77
+
78
+ /**
79
+ * Issue #124: the sample count at which the verification multiplier has
80
+ * fully tapered to 1 (samples >= this are trusted as-is). Between
81
+ * MIN_SHAPE_SAMPLES (full multiplier) and this (1.0) it tapers linearly.
82
+ */
83
+ export const VERIFICATION_MULTIPLIER_FULL_SAMPLES = 6
84
+
85
+ /** CI/GitHub Actions mention in an issue's title/body — a signal that its
86
+ * remediation is verification-heavy (real runner round-trips dominate cost). */
87
+ const CI_MENTION_RE = /(github\s*actions|\bci\b|workflow|runner|continuous\s+integration)/i
88
+
89
+ /**
90
+ * CI/config-change keyword scan (issue #125). Matched against an issue's
91
+ * title AND body; `\b` guards around "ci"/"actions" keep false positives
92
+ * out ("transactions" contains "actions" but is not CI work — the observed
93
+ * 2026-08-17 under-prediction was specifically rounds touching GitHub
94
+ * Actions/workflow config, e.g. w13's `test/hot` group for issues #62/#90/
95
+ * #98 which added .github/workflows and forced real runner round-trips).
96
+ */
97
+ export const CI_CONFIG_RE = /(\bci\b|workflow|\bactions\b|runner)/i
98
+
99
+ /**
100
+ * The flagged higher-bound multiplier for a CI/config-changing issue
101
+ * (issue #125): round-2026-08-17's preflight under-predicted rounds
102
+ * containing CI/test-infrastructure work by 2-3x (~$0.79 estimated vs
103
+ * ~$2.09 actual), so the higher bound uses the top of that observed range.
104
+ */
105
+ export const CI_CONFIG_HIGHER_BOUND_MULTIPLIER = 3
106
+
107
+ /** Whether an issue touches CI/config (title or body — see CI_CONFIG_RE). */
108
+ export function isCiConfigChange(issue: SplitIssue): boolean {
109
+ return CI_CONFIG_RE.test(`${issue.title}\n${issue.body ?? ""}`)
110
+ }
111
+
112
+ /** Per-shape aggregate over recorded history (one per distinct shape seen). */
113
+ export interface ShapeStats {
114
+ shape: string
115
+ /** Distinct recorded groups carrying this shape (a mixed-shape group counts once per shape). */
116
+ samples: number
117
+ /** Min / median / max of the groups' raw combined cost, USD. */
118
+ costMinUsd: number
119
+ costMedianUsd: number
120
+ costMaxUsd: number
121
+ /**
122
+ * Median of costUsd / max(issues.length, 1) — per-issue normalized, since
123
+ * a recorded group can span several issues and a NEW issue is one issue.
124
+ */
125
+ costPerIssueMedianUsd: number
126
+ /** Median of the groups' raw iteration counts. */
127
+ iterationsMedian: number
128
+ /** Mean auto-continuation cycles per group (worker hit the iteration cap and was re-spawned). */
129
+ continuationRate: number
130
+ /** Mean review-rework cycles per group (reviewer found issues, worker was re-spawned to fix them). */
131
+ reworkRate: number
132
+ }
133
+
134
+ /** A single new issue's estimate, matched to recorded history. */
135
+ export interface IssueEstimate {
136
+ issue: number
137
+ shape: string
138
+ /**
139
+ * "direct" — this exact issue number appears in a past record (strongest
140
+ * signal, always usable); "shape" — matched via issueShape, usable only
141
+ * when samples >= MIN_SHAPE_SAMPLES; "none" — no matching history at all.
142
+ */
143
+ match: "direct" | "shape" | "none"
144
+ /** Recorded groups behind this estimate (0 when match === "none"). */
145
+ samples: number
146
+ /** Median normalized per-issue cost, USD. Undefined when not usable / no match. */
147
+ costPerIssueUsd?: number
148
+ /** [min, max] of the same normalized per-issue cost — the range of what a single issue actually cost. */
149
+ costPerIssueRangeUsd?: [number, number]
150
+ /** Median normalized per-issue iterations. */
151
+ iterationsPerIssue?: number
152
+ iterationsPerIssueRange?: [number, number]
153
+ /** Mean continuation cycles per group among the matched records. */
154
+ continuationRate?: number
155
+ /** Mean rework cycles per group among the matched records. */
156
+ reworkRate?: number
157
+ /**
158
+ * Issue #124: multiplier applied to a verification-heavy issue's COST
159
+ * (not iterations) while its shape's samples are thin — see
160
+ * TEST_SHAPE_VERIFICATION_MULTIPLIER. 1 when not verification-heavy,
161
+ * when samples >= VERIFICATION_MULTIPLIER_FULL_SAMPLES, or on a direct
162
+ * match (a direct record of the very task is trusted as-is).
163
+ */
164
+ verificationMultiplier?: number
165
+ /**
166
+ * The issue touches CI/config (title/body matches CI_CONFIG_RE, issue
167
+ * #125). CI work historically costs 2-3x its shape median — see
168
+ * higherBoundUsd.
169
+ */
170
+ ciConfig?: boolean
171
+ /**
172
+ * Flagged higher-bound per-issue cost for a CI/config change:
173
+ * costPerIssueUsd * CI_CONFIG_HIGHER_BOUND_MULTIPLIER. Only present when
174
+ * the issue is a CI/config change AND has usable history (issue #125).
175
+ */
176
+ higherBoundUsd?: number
177
+ }
178
+
179
+ /** A whole dry-run group's estimate — the sum of its issues' per-issue estimates. */
180
+ export interface GroupEstimate {
181
+ name: string
182
+ issues: number[]
183
+ /** Distinct shapes in the group, first-seen order (the dry-run shape label). */
184
+ shapes: string[]
185
+ /** Per-issue estimates, same order as `issues`. */
186
+ perIssue: IssueEstimate[]
187
+ /** Sum of per-issue median cost. Undefined when NO issue has usable history. */
188
+ expectedCostUsd?: number
189
+ /** [sum of per-issue min, sum of per-issue max] — the "rough range" the issue asks for. */
190
+ costRangeUsd?: [number, number]
191
+ /** Sum of per-issue median iterations. */
192
+ expectedIterations?: number
193
+ iterationsRange?: [number, number]
194
+ /** Mean of the per-issue continuation/rework rates (cycles per group). */
195
+ continuationRate?: number
196
+ reworkRate?: number
197
+ /** Issues with no usable history (excluded from the sums above). */
198
+ unmatched: number
199
+ /**
200
+ * Issue #124: the maximum per-issue verification multiplier among the
201
+ * group's USABLE (cost-bearing) estimates — i.e. the largest multiplier
202
+ * that actually scaled the group's expected cost. Undefined when no
203
+ * usable estimate was scaled: a multiplier recorded on an
204
+ * insufficient-data estimate never surfaces here, because it did not
205
+ * scale the displayed cost.
206
+ */
207
+ verificationMultiplier?: number
208
+ /**
209
+ * Issue #125: the flagged higher-bound group total — CI/config-changing
210
+ * issues contribute their higherBoundUsd, everything else its plain
211
+ * median. Only present when at least one usable CI/config issue exists.
212
+ */
213
+ higherBoundUsd?: number
214
+ /**
215
+ * Issue #126: the estimate is built on thin or missing history —
216
+ * unmatched issues (their cost is silently excluded from the total), a
217
+ * usable-but-under-sampled direct match (fewer than MIN_SHAPE_SAMPLES
218
+ * records of the same task), or NO usable history at all. Preflight
219
+ * renders this as a prominent LOW-CONFIDENCE marker.
220
+ */
221
+ lowConfidence: boolean
222
+ }
223
+
224
+ function median(values: number[]): number {
225
+ const sorted = [...values].sort((a, b) => a - b)
226
+ if (sorted.length === 0) {
227
+ return 0
228
+ }
229
+ const mid = Math.floor(sorted.length / 2)
230
+ return sorted.length % 2 === 1 ? (sorted[mid] ?? 0) : ((sorted[mid - 1] ?? 0) + (sorted[mid] ?? 0)) / 2
231
+ }
232
+
233
+ /** min/median/max of a non-empty list (0,0,0 for an empty list). */
234
+ function minMedMax(values: number[]): [number, number, number] {
235
+ if (values.length === 0) {
236
+ return [0, 0, 0]
237
+ }
238
+ return [Math.min(...values), median(values), Math.max(...values)]
239
+ }
240
+
241
+ /** A record's per-issue cost/iterations (a group record can span several issues). */
242
+ function perIssueCost(record: CostHistoryRecord): number {
243
+ return record.costUsd / Math.max(1, (record.issues ?? []).length)
244
+ }
245
+
246
+ function perIssueIterations(record: CostHistoryRecord): number {
247
+ return (record.iterations ?? 0) / Math.max(1, (record.issues ?? []).length)
248
+ }
249
+
250
+ function mean(values: number[]): number {
251
+ if (values.length === 0) {
252
+ return 0
253
+ }
254
+ return values.reduce((sum, v) => sum + v, 0) / values.length
255
+ }
256
+
257
+ /** Summarize a set of same-shape records into ShapeStats. */
258
+ function summarize(shape: string, records: CostHistoryRecord[]): ShapeStats {
259
+ const [costMinUsd, costMedianUsd, costMaxUsd] = minMedMax(records.map((r) => r.costUsd))
260
+ return {
261
+ shape,
262
+ samples: records.length,
263
+ costMinUsd,
264
+ costMedianUsd,
265
+ costMaxUsd,
266
+ costPerIssueMedianUsd: median(records.map(perIssueCost)),
267
+ iterationsMedian: median(records.map((r) => r.iterations ?? 0)),
268
+ continuationRate: mean(records.map((r) => r.continuationCount ?? 0)),
269
+ reworkRate: mean(records.map((r) => r.reworkCount ?? 0)),
270
+ }
271
+ }
272
+
273
+ /**
274
+ * Aggregate recorded history by issue shape. A record with several shapes
275
+ * (a mixed-shape group) contributes to EACH of them — it is evidence about
276
+ * every kind of work it covered. Records with no `shapes` (written before
277
+ * issue #16's field existed) contribute to nothing and are skipped.
278
+ * Returns shapes sorted by sample count (descending), then name.
279
+ */
280
+ export function aggregateByShape(records: CostHistoryRecord[]): ShapeStats[] {
281
+ const byShape = new Map<string, CostHistoryRecord[]>()
282
+ for (const record of records) {
283
+ for (const shape of new Set(record.shapes ?? [])) {
284
+ const list = byShape.get(shape) ?? []
285
+ list.push(record)
286
+ byShape.set(shape, list)
287
+ }
288
+ }
289
+ return [...byShape.entries()]
290
+ .map(([shape, list]) => summarize(shape, list))
291
+ .sort((a, b) => b.samples - a.samples || a.shape.localeCompare(b.shape))
292
+ }
293
+
294
+ /** Whether an issue's work is verification-heavy: test-shaped, or its
295
+ * title/body mentions CI/GitHub Actions (a diff that adds CI config forces
296
+ * real GitHub-runner round-trips, which dominated round-2026-08-17's cost). */
297
+ export function isVerificationHeavy(shape: string, issue?: SplitIssue): boolean {
298
+ if (shape === "test") {
299
+ return true
300
+ }
301
+ if (!issue) {
302
+ return false
303
+ }
304
+ return CI_MENTION_RE.test(`${issue.title}\n${issue.body ?? ""}`)
305
+ }
306
+
307
+ /**
308
+ * Issue #124: the cost multiplier for a verification-heavy issue. Tapers
309
+ * linearly from TEST_SHAPE_VERIFICATION_MULTIPLIER at MIN_SHAPE_SAMPLES to 1
310
+ * at VERIFICATION_MULTIPLIER_FULL_SAMPLES — the thin samples for test/CI
311
+ * shapes are not trusted as-is, but once real samples accumulate the recorded
312
+ * median already reflects real verification cost. CLAMPED at
313
+ * TEST_SHAPE_VERIFICATION_MULTIPLIER below MIN_SHAPE_SAMPLES: a thin-but-
314
+ * nonzero sample set gets the full scale, never more — the multiplier never
315
+ * exceeds its documented maximum (the taper formula is only meaningful on
316
+ * [MIN_SHAPE_SAMPLES, VERIFICATION_MULTIPLIER_FULL_SAMPLES], so it must not
317
+ * be extrapolated below the lower bound). 1 for a direct match (a record of
318
+ * the VERY task is meaningful on its own), for a non-verification-heavy
319
+ * issue, or when no usable numeric estimate is offered (samples <= 0).
320
+ */
321
+ export function verificationMultiplier(
322
+ shape: string,
323
+ issue: SplitIssue | undefined,
324
+ match: IssueEstimate["match"],
325
+ samples: number,
326
+ ): number {
327
+ if (match === "direct" || samples <= 0 || !isVerificationHeavy(shape, issue)) {
328
+ return 1
329
+ }
330
+ if (samples < MIN_SHAPE_SAMPLES) {
331
+ return TEST_SHAPE_VERIFICATION_MULTIPLIER
332
+ }
333
+ if (samples >= VERIFICATION_MULTIPLIER_FULL_SAMPLES) {
334
+ return 1
335
+ }
336
+ const span = VERIFICATION_MULTIPLIER_FULL_SAMPLES - MIN_SHAPE_SAMPLES
337
+ const progress = (samples - MIN_SHAPE_SAMPLES) / span
338
+ return TEST_SHAPE_VERIFICATION_MULTIPLIER - (TEST_SHAPE_VERIFICATION_MULTIPLIER - 1) * progress
339
+ }
340
+
341
+ /** Estimate one new issue from recorded history. */
342
+ export function estimateIssue(
343
+ records: CostHistoryRecord[],
344
+ issueNumber: number,
345
+ shape: string,
346
+ issue?: SplitIssue,
347
+ ): IssueEstimate {
348
+ // Exact issue-number history is the strongest signal — the same task ran
349
+ // before (e.g. a re-opened issue, or an issue batched in a prior round).
350
+ const direct = records.filter((r) => (r.issues ?? []).includes(issueNumber))
351
+ if (direct.length > 0) {
352
+ return issueEstimateFrom(issueNumber, shape, "direct", direct, issue)
353
+ }
354
+ const byShape = records.filter((r) => (r.shapes ?? []).includes(shape))
355
+ if (byShape.length >= MIN_SHAPE_SAMPLES) {
356
+ return issueEstimateFrom(issueNumber, shape, "shape", byShape, issue)
357
+ }
358
+ return {
359
+ issue: issueNumber,
360
+ shape,
361
+ match: byShape.length > 0 ? "shape" : "none",
362
+ samples: byShape.length,
363
+ verificationMultiplier: verificationMultiplier(shape, issue, "shape", byShape.length),
364
+ }
365
+ }
366
+
367
+ function issueEstimateFrom(
368
+ issueNumber: number,
369
+ shape: string,
370
+ match: IssueEstimate["match"],
371
+ records: CostHistoryRecord[],
372
+ issue?: SplitIssue,
373
+ ): IssueEstimate {
374
+ const [costLo, costMed, costHi] = minMedMax(records.map(perIssueCost))
375
+ const [iterLo, iterMed, iterHi] = minMedMax(records.map(perIssueIterations))
376
+ const multiplier = verificationMultiplier(shape, issue, match, records.length)
377
+ const costPerIssueUsd = costMed * multiplier
378
+ const estimate: IssueEstimate = {
379
+ issue: issueNumber,
380
+ shape,
381
+ match,
382
+ samples: records.length,
383
+ costPerIssueUsd,
384
+ costPerIssueRangeUsd: [costLo * multiplier, costHi * multiplier],
385
+ iterationsPerIssue: iterMed,
386
+ iterationsPerIssueRange: [iterLo, iterHi],
387
+ continuationRate: mean(records.map((r) => r.continuationCount ?? 0)),
388
+ reworkRate: mean(records.map((r) => r.reworkCount ?? 0)),
389
+ verificationMultiplier: multiplier,
390
+ }
391
+ // Issue #125: flag the CI/config issue and stamp its higher bound (the
392
+ // per-issue median, multiplied) — a separate, complementary signal to
393
+ // the verification multiplier above, which scales the primary estimate;
394
+ // this instead surfaces an additional flagged higher-bound alongside it.
395
+ if (issue !== undefined && isCiConfigChange(issue)) {
396
+ estimate.ciConfig = true
397
+ estimate.higherBoundUsd = costPerIssueUsd * CI_CONFIG_HIGHER_BOUND_MULTIPLIER
398
+ }
399
+ return estimate
400
+ }
401
+
402
+ function sum(values: number[]): number {
403
+ return values.reduce((acc, v) => acc + v, 0)
404
+ }
405
+
406
+ /** Combine a spec's per-issue estimates into a group-level estimate. */
407
+ export function estimateGroup(spec: WorktreeSpec, perIssue: IssueEstimate[]): GroupEstimate {
408
+ const usable = perIssue.filter((e) => e.costPerIssueUsd !== undefined && e.iterationsPerIssue !== undefined)
409
+ const costs = usable.map((e) => e.costPerIssueUsd ?? 0)
410
+ const iters = usable.map((e) => e.iterationsPerIssue ?? 0)
411
+ const withRates = usable.filter((e) => e.continuationRate !== undefined && e.reworkRate !== undefined)
412
+ const costLo = sum(usable.map((e) => e.costPerIssueRangeUsd?.[0] ?? 0))
413
+ const costHi = sum(usable.map((e) => e.costPerIssueRangeUsd?.[1] ?? 0))
414
+ const iterLo = sum(usable.map((e) => e.iterationsPerIssueRange?.[0] ?? 0))
415
+ const iterHi = sum(usable.map((e) => e.iterationsPerIssueRange?.[1] ?? 0))
416
+ const estimate: GroupEstimate = {
417
+ name: spec.name,
418
+ issues: spec.issues,
419
+ shapes: [...new Set(perIssue.map((e) => e.shape))],
420
+ perIssue,
421
+ unmatched: perIssue.length - usable.length,
422
+ // Issue #126: a usable DIRECT match with fewer than MIN_SHAPE_SAMPLES
423
+ // records of the same task is still an anecdote, not a real estimate
424
+ // (estimateIssue offers it because one record of the very task IS
425
+ // meaningful, but its confidence is low).
426
+ lowConfidence: perIssue.some(
427
+ (e) => e.costPerIssueUsd === undefined || (e.match === "direct" && e.samples < MIN_SHAPE_SAMPLES),
428
+ ),
429
+ }
430
+ // Only a multiplier that actually scaled a USABLE estimate (one that
431
+ // contributes to expectedCostUsd) is surfaced — a multiplier recorded on
432
+ // an insufficient-data estimate never scaled the displayed cost, so
433
+ // claiming it in the dry-run line would be false.
434
+ const multipliers = usable
435
+ .map((e) => e.verificationMultiplier ?? 1)
436
+ .filter((m) => m > 1)
437
+ if (multipliers.length > 0) {
438
+ estimate.verificationMultiplier = Math.max(...multipliers)
439
+ }
440
+ // Issue #125: flagged higher-bound total — CI/config-changing issues
441
+ // historically cost 2-3x their shape median (round-2026-08-17: w13
442
+ // `test/hot` estimated ~$0.05-$0.27, actual $1.03 then $1.11 with
443
+ // review). Emit it whenever at least one usable CI/config issue exists.
444
+ const ciUsable = usable.filter((e) => e.ciConfig === true)
445
+ if (ciUsable.length > 0) {
446
+ estimate.higherBoundUsd = sum(
447
+ usable.map((e) =>
448
+ e.ciConfig === true
449
+ ? (e.higherBoundUsd ?? e.costPerIssueUsd ?? 0)
450
+ : (e.costPerIssueUsd ?? 0),
451
+ ),
452
+ )
453
+ }
454
+ if (usable.length > 0) {
455
+ estimate.expectedCostUsd = sum(costs)
456
+ estimate.costRangeUsd = [costLo, costHi]
457
+ estimate.expectedIterations = sum(iters)
458
+ estimate.iterationsRange = [iterLo, iterHi]
459
+ }
460
+ if (withRates.length > 0) {
461
+ estimate.continuationRate = mean(withRates.map((e) => e.continuationRate ?? 0))
462
+ estimate.reworkRate = mean(withRates.map((e) => e.reworkRate ?? 0))
463
+ }
464
+ return estimate
465
+ }
466
+
467
+ /**
468
+ * Estimate every dry-run group from recorded history. Shapes are computed
469
+ * from the loaded issues (the only place titles/bodies exist for NEW work);
470
+ * a number without a loaded issue falls back to "generic".
471
+ */
472
+ export function estimateGroups(records: CostHistoryRecord[], specs: WorktreeSpec[], issues: SplitIssue[]): GroupEstimate[] {
473
+ const shapes = new Map<number, string>()
474
+ for (const issue of issues) {
475
+ shapes.set(issue.number, issueShape(issue))
476
+ }
477
+ return specs.map((spec) => {
478
+ const perIssue = spec.issues.map((n) => estimateIssue(records, n, shapes.get(n) ?? "generic", issues.find((i) => i.number === n)))
479
+ return estimateGroup(spec, perIssue)
480
+ })
481
+ }
482
+
483
+ function formatUsd(value: number): string {
484
+ return `$${value.toFixed(4)}`
485
+ }
486
+
487
+ function formatRate(value: number | undefined): string {
488
+ return value === undefined ? "?" : value.toFixed(1)
489
+ }
490
+
491
+ /**
492
+ * Human-readable dry-run section lines (one per group, "rough" by design).
493
+ * `historyRecordCount` is the total recorded groups; 0 means the section is
494
+ * a single "no history yet" note rather than N identical per-group ones.
495
+ */
496
+ export function buildEstimateSection(estimates: GroupEstimate[], historyRecordCount: number): string[] {
497
+ if (historyRecordCount === 0) {
498
+ return ["no cost history recorded yet — estimates appear once groups reach a terminal status (cost-history.ts)"]
499
+ }
500
+ const lines: string[] = []
501
+ for (const group of estimates) {
502
+ const issuesLabel = group.issues.join(", ")
503
+ const shapeLabel = group.shapes.join("/")
504
+ if (group.expectedCostUsd !== undefined && group.costRangeUsd && group.expectedIterations !== undefined) {
505
+ const [costLo, costHi] = group.costRangeUsd
506
+ const [iterLo, iterHi] = group.iterationsRange ?? [group.expectedIterations, group.expectedIterations]
507
+ const samples = Math.max(0, ...group.perIssue.map((e) => e.samples))
508
+ let line =
509
+ `${group.name} (issues ${issuesLabel} — ${shapeLabel}): expected ~${formatUsd(costLo)}–${formatUsd(costHi)} · ` +
510
+ `~${iterLo}–${iterHi} iterations · ${samples} sample(s)`
511
+ if (group.verificationMultiplier !== undefined && group.verificationMultiplier > 1) {
512
+ line += ` · ${group.verificationMultiplier.toFixed(2)}x verification multiplier`
513
+ }
514
+ if (group.unmatched > 0) {
515
+ line += ` (${group.unmatched} issue(s) unmatched)`
516
+ }
517
+ // Issue #125: CI/config-changing issues historically cost 2-3x
518
+ // their shape median — surface the flagged higher bound right on
519
+ // the estimate line instead of burying it in a footnote.
520
+ if (group.higherBoundUsd !== undefined) {
521
+ line += ` · CI/config work: higher bound ~${formatUsd(group.higherBoundUsd)}`
522
+ }
523
+ lines.push(line)
524
+ lines.push(
525
+ ` continuation/rework: ${formatRate(group.continuationRate)} / ${formatRate(group.reworkRate)} cycles per group avg`,
526
+ )
527
+ } else {
528
+ const withData = group.perIssue.filter((e) => e.match !== "none")
529
+ if (withData.length > 0) {
530
+ const samples = Math.max(...withData.map((e) => e.samples))
531
+ lines.push(
532
+ `${group.name} (issues ${issuesLabel} — ${shapeLabel}): insufficient data (${samples} sample(s), need ${MIN_SHAPE_SAMPLES}) — no estimate yet`,
533
+ )
534
+ } else {
535
+ lines.push(`${group.name} (issues ${issuesLabel} — ${shapeLabel}): no history for this shape yet`)
536
+ }
537
+ }
538
+ // Issue #126: prominent LOW-CONFIDENCE marker whenever the estimate
539
+ // is built on thin or missing history (unmatched issues, an
540
+ // under-sampled direct match, or no usable history at all).
541
+ if (group.lowConfidence) {
542
+ lines.push(` ⚠ LOW-CONFIDENCE: ${lowConfidenceReason(group)}`)
543
+ }
544
+ }
545
+ return lines
546
+ }
547
+
548
+ /** Human-readable reason behind a group's low-confidence flag (issue #126). */
549
+ export function lowConfidenceReason(group: GroupEstimate): string {
550
+ if (group.expectedCostUsd === undefined) {
551
+ const withData = group.perIssue.filter((e) => e.match !== "none")
552
+ return withData.length > 0
553
+ ? `insufficient history (${Math.max(...withData.map((e) => e.samples))} sample(s), need ${MIN_SHAPE_SAMPLES}) — no numeric estimate`
554
+ : "no history for this shape yet"
555
+ }
556
+ if (group.unmatched > 0) {
557
+ return `${group.unmatched} issue(s) unmatched (their cost is NOT included in the estimate)`
558
+ }
559
+ const underSampled = group.perIssue.find((e) => e.match === "direct" && e.samples < MIN_SHAPE_SAMPLES)
560
+ if (underSampled) {
561
+ return `direct history for issue ${underSampled.issue} is a single-run anecdote (${underSampled.samples} sample(s), need ${MIN_SHAPE_SAMPLES})`
562
+ }
563
+ return "estimate relies on thin history"
564
+ }