forge-workflow 0.0.4 → 0.0.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (248) hide show
  1. package/.claude/commands/dev.md +345 -340
  2. package/.claude/commands/plan.md +566 -521
  3. package/.claude/commands/premerge.md +186 -176
  4. package/.claude/commands/research.md +42 -42
  5. package/.claude/commands/review.md +448 -442
  6. package/.claude/commands/rollback.md +721 -721
  7. package/.claude/commands/ship.md +212 -164
  8. package/.claude/commands/sonarcloud.md +152 -152
  9. package/.claude/commands/status.md +90 -48
  10. package/.claude/commands/validate.md +288 -282
  11. package/.claude/commands/verify.md +269 -221
  12. package/.claude/rules/greptile-review-process.md +285 -285
  13. package/.claude/rules/workflow.md +121 -105
  14. package/.claude/scripts/greptile-resolve.sh +558 -526
  15. package/.claude/scripts/load-env.sh +32 -32
  16. package/.cline/workflows/dev.md +342 -337
  17. package/.cline/workflows/plan.md +563 -518
  18. package/.cline/workflows/premerge.md +183 -173
  19. package/.cline/workflows/research.md +39 -39
  20. package/.cline/workflows/review.md +445 -439
  21. package/.cline/workflows/rollback.md +718 -718
  22. package/.cline/workflows/ship.md +209 -161
  23. package/.cline/workflows/sonarcloud.md +146 -146
  24. package/.cline/workflows/status.md +87 -45
  25. package/.cline/workflows/validate.md +285 -279
  26. package/.cline/workflows/verify.md +266 -218
  27. package/.codex/config.toml +11 -11
  28. package/.codex/skills/dev/SKILL.md +345 -340
  29. package/.codex/skills/plan/SKILL.md +566 -521
  30. package/.codex/skills/premerge/SKILL.md +186 -176
  31. package/.codex/skills/research/SKILL.md +42 -42
  32. package/.codex/skills/review/SKILL.md +448 -442
  33. package/.codex/skills/rollback/SKILL.md +721 -721
  34. package/.codex/skills/ship/SKILL.md +212 -164
  35. package/.codex/skills/sonarcloud/SKILL.md +149 -149
  36. package/.codex/skills/status/SKILL.md +90 -48
  37. package/.codex/skills/validate/SKILL.md +288 -282
  38. package/.codex/skills/verify/SKILL.md +269 -221
  39. package/.cursor/commands/dev.md +342 -337
  40. package/.cursor/commands/plan.md +563 -518
  41. package/.cursor/commands/premerge.md +183 -173
  42. package/.cursor/commands/research.md +39 -39
  43. package/.cursor/commands/review.md +445 -439
  44. package/.cursor/commands/rollback.md +718 -718
  45. package/.cursor/commands/ship.md +209 -161
  46. package/.cursor/commands/sonarcloud.md +146 -146
  47. package/.cursor/commands/status.md +87 -45
  48. package/.cursor/commands/validate.md +285 -279
  49. package/.cursor/commands/verify.md +266 -218
  50. package/.cursor/rules/permissions-guidance.mdc +37 -37
  51. package/.forge/hooks/check-tdd.js +240 -240
  52. package/.github/PLUGIN_TEMPLATE.json +32 -32
  53. package/.github/prompts/dev.prompt.md +347 -342
  54. package/.github/prompts/plan.prompt.md +568 -523
  55. package/.github/prompts/premerge.prompt.md +188 -178
  56. package/.github/prompts/research.prompt.md +44 -44
  57. package/.github/prompts/review.prompt.md +450 -444
  58. package/.github/prompts/rollback.prompt.md +723 -723
  59. package/.github/prompts/ship.prompt.md +214 -166
  60. package/.github/prompts/sonarcloud.prompt.md +151 -151
  61. package/.github/prompts/status.prompt.md +92 -50
  62. package/.github/prompts/validate.prompt.md +290 -284
  63. package/.github/prompts/verify.prompt.md +271 -223
  64. package/.github/workflows/beads-to-github.yml +56 -0
  65. package/.github/workflows/github-to-beads.yml +97 -0
  66. package/.kilocode/workflows/dev.md +346 -341
  67. package/.kilocode/workflows/plan.md +567 -522
  68. package/.kilocode/workflows/premerge.md +187 -177
  69. package/.kilocode/workflows/research.md +43 -43
  70. package/.kilocode/workflows/review.md +449 -443
  71. package/.kilocode/workflows/rollback.md +722 -722
  72. package/.kilocode/workflows/ship.md +213 -165
  73. package/.kilocode/workflows/sonarcloud.md +150 -150
  74. package/.kilocode/workflows/status.md +91 -49
  75. package/.kilocode/workflows/validate.md +289 -283
  76. package/.kilocode/workflows/verify.md +270 -222
  77. package/.mcp.json.example +12 -12
  78. package/.opencode/commands/dev.md +345 -340
  79. package/.opencode/commands/plan.md +566 -521
  80. package/.opencode/commands/premerge.md +186 -176
  81. package/.opencode/commands/research.md +42 -42
  82. package/.opencode/commands/review.md +448 -442
  83. package/.opencode/commands/rollback.md +721 -721
  84. package/.opencode/commands/ship.md +212 -164
  85. package/.opencode/commands/sonarcloud.md +149 -149
  86. package/.opencode/commands/status.md +90 -48
  87. package/.opencode/commands/validate.md +288 -282
  88. package/.opencode/commands/verify.md +269 -221
  89. package/.roo/commands/dev.md +346 -341
  90. package/.roo/commands/plan.md +567 -522
  91. package/.roo/commands/premerge.md +187 -177
  92. package/.roo/commands/research.md +43 -43
  93. package/.roo/commands/review.md +449 -443
  94. package/.roo/commands/rollback.md +722 -722
  95. package/.roo/commands/ship.md +213 -165
  96. package/.roo/commands/sonarcloud.md +150 -150
  97. package/.roo/commands/status.md +91 -49
  98. package/.roo/commands/validate.md +289 -283
  99. package/.roo/commands/verify.md +270 -222
  100. package/AGENTS.md +272 -175
  101. package/CLAUDE.md +110 -100
  102. package/README.md +429 -416
  103. package/bin/forge-cmd.js +317 -313
  104. package/bin/forge-preflight.js +322 -309
  105. package/bin/forge.js +4765 -4303
  106. package/docs/AGENT_INSTALL_PROMPT.md +342 -342
  107. package/docs/BEADS_GITHUB_SYNC.md +251 -251
  108. package/docs/ENHANCED_ONBOARDING.md +612 -602
  109. package/docs/EXAMPLES.md +482 -482
  110. package/docs/GREPTILE_SETUP.md +400 -400
  111. package/docs/MANUAL_REVIEW_GUIDE.md +106 -106
  112. package/docs/ROADMAP.md +359 -359
  113. package/docs/SETUP.md +663 -631
  114. package/docs/TOOLCHAIN.md +653 -630
  115. package/docs/VALIDATION.md +363 -363
  116. package/install.sh +40 -1056
  117. package/lefthook.yml +50 -39
  118. package/lib/agents/README.md +198 -198
  119. package/lib/agents/claude.plugin.json +28 -28
  120. package/lib/agents/cline.plugin.json +22 -22
  121. package/lib/agents/codex.plugin.json +19 -19
  122. package/lib/agents/copilot.plugin.json +24 -24
  123. package/lib/agents/cursor.plugin.json +25 -25
  124. package/lib/agents/kilocode.plugin.json +22 -22
  125. package/lib/agents/opencode.plugin.json +20 -20
  126. package/lib/agents/roo.plugin.json +23 -23
  127. package/lib/agents-config.js +2112 -2112
  128. package/lib/beads-health-check.js +143 -0
  129. package/lib/beads-setup.js +341 -0
  130. package/lib/beads-sync-scaffold.js +260 -0
  131. package/lib/commands/_registry.js +134 -0
  132. package/lib/commands/clean.js +181 -0
  133. package/lib/commands/dev.js +571 -513
  134. package/lib/commands/plan.js +692 -692
  135. package/lib/commands/push.js +196 -0
  136. package/lib/commands/recommend.js +119 -119
  137. package/lib/commands/ship.js +377 -377
  138. package/lib/commands/status.js +378 -378
  139. package/lib/commands/sync.js +55 -0
  140. package/lib/commands/team.js +37 -0
  141. package/lib/commands/test.js +207 -0
  142. package/lib/commands/validate.js +602 -602
  143. package/lib/commands/worktree.js +310 -0
  144. package/lib/context-merge.js +359 -359
  145. package/lib/dep-guard/analyzer.js +294 -294
  146. package/lib/dep-guard/behavior-detector.js +98 -98
  147. package/lib/dep-guard/contract-detector.js +162 -162
  148. package/lib/dep-guard/import-detector.js +498 -498
  149. package/lib/dep-guard/path-utils.js +13 -13
  150. package/lib/dep-guard/rubric.js +120 -120
  151. package/lib/dep-guard/task-parser.js +318 -318
  152. package/lib/detect-agent.js +191 -191
  153. package/lib/detect-worktree.js +47 -47
  154. package/lib/docs-command.js +51 -0
  155. package/lib/docs-copy.js +50 -0
  156. package/lib/file-hash.js +26 -26
  157. package/lib/freshness-token.js +148 -0
  158. package/lib/greptile-match.js +80 -0
  159. package/lib/husky-migration.js +450 -0
  160. package/lib/lefthook-check.js +65 -0
  161. package/lib/pat-setup.js +207 -0
  162. package/lib/plugin-catalog.js +350 -350
  163. package/lib/plugin-manager.js +166 -166
  164. package/lib/plugin-recommender.js +141 -141
  165. package/lib/project-discovery.js +491 -491
  166. package/lib/reset.js +309 -0
  167. package/lib/setup-action-log.js +139 -139
  168. package/lib/setup-summary-renderer.js +106 -106
  169. package/lib/setup-utils.js +96 -0
  170. package/lib/setup.js +192 -192
  171. package/lib/smart-merge.js +64 -0
  172. package/lib/symlink-utils.js +81 -0
  173. package/lib/task-ownership.js +117 -0
  174. package/lib/workflow-profiles.js +197 -197
  175. package/package.json +131 -128
  176. package/scripts/beads-context.sh +426 -0
  177. package/scripts/beads-context.test.js +567 -0
  178. package/scripts/behavioral-judge.sh +378 -0
  179. package/scripts/benchmark.js +85 -0
  180. package/scripts/branch-protection.js +183 -0
  181. package/scripts/check-agents.js +172 -0
  182. package/scripts/check-forge-token.js +98 -0
  183. package/scripts/commitlint.js +42 -0
  184. package/scripts/conflict-detect.sh +323 -0
  185. package/scripts/dep-guard-analyze.js +71 -0
  186. package/scripts/dep-guard.sh +789 -0
  187. package/scripts/eval_win.py +249 -0
  188. package/scripts/file-index.sh +493 -0
  189. package/scripts/forge-team/index.sh +86 -0
  190. package/scripts/forge-team/lib/agent-prompt.sh +52 -0
  191. package/scripts/forge-team/lib/claim.sh +256 -0
  192. package/scripts/forge-team/lib/dashboard.sh +341 -0
  193. package/scripts/forge-team/lib/epic.sh +332 -0
  194. package/scripts/forge-team/lib/hooks.sh +253 -0
  195. package/scripts/forge-team/lib/identity.sh +235 -0
  196. package/scripts/forge-team/lib/sync-github.sh +317 -0
  197. package/scripts/forge-team/lib/verify.sh +284 -0
  198. package/scripts/forge-team/lib/workload.sh +296 -0
  199. package/scripts/forge-team/tests/agent-prompt.test.sh +72 -0
  200. package/scripts/forge-team/tests/claim.test.sh +179 -0
  201. package/scripts/forge-team/tests/dashboard.test.sh +170 -0
  202. package/scripts/forge-team/tests/dispatcher.test.sh +79 -0
  203. package/scripts/forge-team/tests/epic.test.sh +176 -0
  204. package/scripts/forge-team/tests/hooks.test.sh +239 -0
  205. package/scripts/forge-team/tests/identity.test.sh +176 -0
  206. package/scripts/forge-team/tests/integration.test.sh +371 -0
  207. package/scripts/forge-team/tests/sync-github.test.sh +209 -0
  208. package/scripts/forge-team/tests/verify.test.sh +314 -0
  209. package/scripts/forge-team/tests/workflow-integration.test.sh +43 -0
  210. package/scripts/forge-team/tests/workload.test.sh +209 -0
  211. package/scripts/github-beads-sync/comment.mjs +64 -0
  212. package/scripts/github-beads-sync/config.mjs +148 -0
  213. package/scripts/github-beads-sync/github-api.mjs +131 -0
  214. package/scripts/github-beads-sync/index.mjs +332 -0
  215. package/scripts/github-beads-sync/label-mapper.mjs +54 -0
  216. package/scripts/github-beads-sync/mapping.mjs +78 -0
  217. package/scripts/github-beads-sync/reverse-sync-cli.mjs +31 -0
  218. package/scripts/github-beads-sync/reverse-sync.mjs +138 -0
  219. package/scripts/github-beads-sync/run-bd.mjs +159 -0
  220. package/scripts/github-beads-sync/sanitize.mjs +121 -0
  221. package/scripts/github-beads-sync.config.json +26 -0
  222. package/scripts/improve-command.js +375 -0
  223. package/scripts/lib/eval-runner.js +268 -0
  224. package/scripts/lib/eval-schema.js +135 -0
  225. package/scripts/lib/eval-storage.js +78 -0
  226. package/scripts/lib/grading.js +203 -0
  227. package/scripts/lib/jsonl-lock.sh +48 -0
  228. package/scripts/lib/sanitize.sh +116 -0
  229. package/scripts/lib/transcript-parser.js +63 -0
  230. package/scripts/lint.js +47 -0
  231. package/scripts/migrate-to-bun-test.js +412 -0
  232. package/scripts/pr-coordinator.sh +706 -0
  233. package/scripts/run-command-eval.js +236 -0
  234. package/scripts/smart-status.sh +809 -0
  235. package/scripts/sync-commands.js +571 -0
  236. package/scripts/sync-utils.sh +455 -0
  237. package/scripts/test-dashboard.js +123 -0
  238. package/scripts/test.js +46 -0
  239. package/scripts/validate.sh +94 -0
  240. package/skills/parallel-deep-research/SKILL.md +108 -108
  241. package/skills/parallel-deep-research/evals/README.md +27 -27
  242. package/skills/parallel-deep-research/evals/evals.json +62 -62
  243. package/skills/sonarcloud-analysis/SKILL.md +171 -171
  244. package/skills/sonarcloud-analysis/evals/README.md +27 -27
  245. package/skills/sonarcloud-analysis/evals/evals.json +50 -50
  246. package/skills/sonarcloud-analysis/references/api-reference.md +466 -466
  247. package/.cursor/hooks/state/continual-learning-index.json +0 -19
  248. package/.cursor/hooks/state/continual-learning.json +0 -8
@@ -0,0 +1,375 @@
1
+ /**
2
+ * Semi-autonomous improvement loop: analyze eval failures, rewrite command,
3
+ * re-evaluate, and stop on regression or plateau.
4
+ *
5
+ * Usage:
6
+ * bun scripts/improve-command.js <command-path> --eval-set <path> [--max-iterations <N>]
7
+ */
8
+
9
+ const { execFileSync } = require('child_process');
10
+ const fs = require('fs');
11
+ const { DEFAULT_BASE_PATH, loadEvalHistory } = require('./lib/eval-storage');
12
+
13
+ // ---------------------------------------------------------------------------
14
+ // analyzeFailures
15
+ // ---------------------------------------------------------------------------
16
+
17
+ /**
18
+ * Extract failing assertions with their query context from an eval result.
19
+ *
20
+ * @param {object} evalResult - result from runEvalPipeline
21
+ * @returns {Array<{ query: string, assertion: string, reasoning: string }>}
22
+ */
23
+ function analyzeFailures(evalResult) {
24
+ const failures = [];
25
+
26
+ for (const queryResult of evalResult.results) {
27
+ for (const assertion of queryResult.assertions) {
28
+ if (!assertion.pass) {
29
+ failures.push({
30
+ query: queryResult.prompt,
31
+ assertion: assertion.check,
32
+ reasoning: assertion.reasoning,
33
+ });
34
+ }
35
+ }
36
+ }
37
+
38
+ return failures;
39
+ }
40
+
41
+ // ---------------------------------------------------------------------------
42
+ // buildRewritePrompt
43
+ // ---------------------------------------------------------------------------
44
+
45
+ /**
46
+ * Build a prompt asking for a command rewrite that fixes the identified failures.
47
+ *
48
+ * @param {string} commandContent - current command markdown content
49
+ * @param {Array<{ query: string, assertion: string, reasoning: string }>} failures
50
+ * @param {object[]} history - array of prior eval results (from loadEvalHistory)
51
+ * @returns {string}
52
+ */
53
+ function buildRewritePrompt(commandContent, failures, history) {
54
+ let prompt = '';
55
+
56
+ prompt += 'You are a command prompt engineer. Rewrite the following command to fix the failing assertions.\n\n';
57
+
58
+ prompt += '## Current Command\n\n';
59
+ prompt += commandContent + '\n\n';
60
+
61
+ prompt += '## Failing Assertions\n\n';
62
+ for (const failure of failures) {
63
+ prompt += `- Query: "${failure.query}"\n`;
64
+ prompt += ` Assertion: "${failure.assertion}"\n`;
65
+ prompt += ` Reasoning: "${failure.reasoning}"\n\n`;
66
+ }
67
+
68
+ if (history && history.length > 0) {
69
+ prompt += '## Prior Eval Attempts\n\n';
70
+ prompt += 'Avoid repeating approaches that have already been tried. Here is a summary of prior attempts:\n\n';
71
+ for (let i = 0; i < history.length; i++) {
72
+ const attempt = history[i];
73
+ const failCount = attempt.results
74
+ ? attempt.results.filter((result) => result.assertions && result.assertions.some((assertion) => !assertion.pass)).length
75
+ : 0;
76
+ prompt += `- Attempt ${i + 1}: score ${attempt.overall_score}, ${failCount} failing queries\n`;
77
+ }
78
+ prompt += '\n';
79
+
80
+ // Detect flaky assertions: the same check both passed and failed across sessions.
81
+ const assertionOutcomes = new Map();
82
+ for (const attempt of history) {
83
+ if (!attempt.results) continue;
84
+ for (const result of attempt.results) {
85
+ if (!result.assertions) continue;
86
+ for (const assertion of result.assertions) {
87
+ if (!assertionOutcomes.has(assertion.check)) {
88
+ assertionOutcomes.set(assertion.check, { pass: 0, fail: 0 });
89
+ }
90
+ const entry = assertionOutcomes.get(assertion.check);
91
+ if (assertion.pass) {
92
+ entry.pass++;
93
+ } else {
94
+ entry.fail++;
95
+ }
96
+ }
97
+ }
98
+ }
99
+
100
+ const flakyAssertions = [];
101
+ for (const [check, outcomes] of assertionOutcomes) {
102
+ if (outcomes.pass > 0 && outcomes.fail > 0) {
103
+ flakyAssertions.push({ check, pass: outcomes.pass, fail: outcomes.fail });
104
+ }
105
+ }
106
+
107
+ if (flakyAssertions.length > 0) {
108
+ prompt += '## Flaky/Inconsistent Assertions\n\n';
109
+ prompt += 'These assertions are flaky - they pass in some sessions and fail in others. ';
110
+ prompt += 'Do not waste iterations on these; they may depend on environment rather than command quality.\n\n';
111
+ for (const flakyAssertion of flakyAssertions) {
112
+ prompt += `- "${flakyAssertion.check}" - passed ${flakyAssertion.pass}x, failed ${flakyAssertion.fail}x across sessions\n`;
113
+ }
114
+ prompt += '\n';
115
+ }
116
+ }
117
+
118
+ prompt += '## Instructions\n\n';
119
+ prompt += 'Return ONLY the rewritten command markdown content. No explanation, no code fences.\n';
120
+
121
+ return prompt;
122
+ }
123
+
124
+ // ---------------------------------------------------------------------------
125
+ // generateDiff
126
+ // ---------------------------------------------------------------------------
127
+
128
+ /**
129
+ * Simple line-by-line diff showing added (+) and removed (-) lines.
130
+ *
131
+ * @param {string} original
132
+ * @param {string} modified
133
+ * @returns {string}
134
+ */
135
+ function generateDiff(original, modified) {
136
+ const originalLines = original.split('\n');
137
+ const modifiedLines = modified.split('\n');
138
+
139
+ if (original === modified) {
140
+ return '';
141
+ }
142
+
143
+ const lines = [];
144
+ let i = 0;
145
+ let j = 0;
146
+
147
+ while (i < originalLines.length || j < modifiedLines.length) {
148
+ if (i < originalLines.length && j < modifiedLines.length && originalLines[i] === modifiedLines[j]) {
149
+ lines.push(' ' + originalLines[i]);
150
+ i++;
151
+ j++;
152
+ } else {
153
+ const originalLineInModified = j < modifiedLines.length ? modifiedLines.indexOf(originalLines[i], j) : -1;
154
+ const modifiedLineInOriginal = i < originalLines.length ? originalLines.indexOf(modifiedLines[j], i) : -1;
155
+
156
+ if (i >= originalLines.length) {
157
+ lines.push('+' + modifiedLines[j]);
158
+ j++;
159
+ } else if (j >= modifiedLines.length) {
160
+ lines.push('-' + originalLines[i]);
161
+ i++;
162
+ } else if (
163
+ originalLineInModified !== -1 &&
164
+ (modifiedLineInOriginal === -1 || originalLineInModified - j <= modifiedLineInOriginal - i)
165
+ ) {
166
+ lines.push('+' + modifiedLines[j]);
167
+ j++;
168
+ } else {
169
+ lines.push('-' + originalLines[i]);
170
+ i++;
171
+ }
172
+ }
173
+ }
174
+
175
+ return lines.join('\n');
176
+ }
177
+
178
+ /**
179
+ * Default rewrite invocation via `claude -p`.
180
+ *
181
+ * @param {string} prompt
182
+ * @param {{ timeout?: number, _execFileSync?: Function }} [options]
183
+ * @returns {Promise<string>}
184
+ */
185
+ async function defaultRewriteCommand(prompt, options = {}) {
186
+ const timeout = options.timeout || 120_000;
187
+ const execFile = options._execFileSync || execFileSync;
188
+
189
+ return execFile(
190
+ 'claude',
191
+ ['-p', prompt, '--output-format', 'text', '--no-session-persistence'],
192
+ {
193
+ encoding: 'utf-8',
194
+ timeout,
195
+ maxBuffer: 10 * 1024 * 1024,
196
+ }
197
+ );
198
+ }
199
+
200
+ // ---------------------------------------------------------------------------
201
+ // runImprovementLoop
202
+ // ---------------------------------------------------------------------------
203
+
204
+ /**
205
+ * Main improvement loop orchestrator.
206
+ *
207
+ * @param {string} commandPath - path to the command markdown file
208
+ * @param {string} evalSetPath - path to the .eval.json file
209
+ * @param {object} [options]
210
+ * @param {number} [options.maxIterations=3]
211
+ * @param {Function} [options._runEval] - injectable eval function for testing
212
+ * @param {Function} [options._rewriteCommand] - injectable rewriter for testing
213
+ * @param {string} [options._basePath] - eval-logs base path for testing
214
+ * @returns {Promise<{ original: string, best: string, originalScore: number, bestScore: number, iterations: number, reason: string, diff: string }>}
215
+ */
216
+ async function runImprovementLoop(commandPath, evalSetPath, options = {}) {
217
+ const maxIterations = options.maxIterations || 3;
218
+ const runEval = options._runEval;
219
+ const rewriteCommand = options._rewriteCommand || ((prompt) => defaultRewriteCommand(prompt, options));
220
+ const basePath = options._basePath || DEFAULT_BASE_PATH;
221
+
222
+ const originalContent = fs.readFileSync(commandPath, 'utf8');
223
+ let bestContent = originalContent;
224
+
225
+ try {
226
+ const baselineResult = await runEval(evalSetPath);
227
+ const originalScore = baselineResult.overall_score;
228
+ const history = loadEvalHistory(baselineResult.command, basePath);
229
+
230
+ let bestScore = originalScore;
231
+ let latestResult = baselineResult;
232
+ let previousScore = originalScore;
233
+ let iterations = 0;
234
+ let reason = 'max_iterations';
235
+
236
+ for (let iter = 1; iter <= maxIterations; iter++) {
237
+ iterations = iter;
238
+
239
+ const failures = analyzeFailures(latestResult);
240
+ const prompt = buildRewritePrompt(
241
+ fs.readFileSync(commandPath, 'utf8'),
242
+ failures,
243
+ history
244
+ );
245
+
246
+ const newContent = await rewriteCommand(prompt);
247
+ fs.writeFileSync(commandPath, newContent, 'utf8');
248
+
249
+ const newResult = await runEval(evalSetPath);
250
+ const newScore = newResult.overall_score;
251
+
252
+ if (newScore < bestScore) {
253
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
254
+ reason = 'regression';
255
+ break;
256
+ }
257
+
258
+ if (newScore === previousScore) {
259
+ if (newScore >= bestScore) {
260
+ bestContent = newContent;
261
+ bestScore = newScore;
262
+ }
263
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
264
+ reason = 'plateau';
265
+ break;
266
+ }
267
+
268
+ if (newScore > bestScore) {
269
+ bestContent = newContent;
270
+ bestScore = newScore;
271
+ }
272
+
273
+ previousScore = newScore;
274
+ latestResult = newResult;
275
+
276
+ if (iter === maxIterations) {
277
+ reason = 'max_iterations';
278
+ }
279
+ }
280
+
281
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
282
+
283
+ return {
284
+ original: originalContent,
285
+ best: bestContent,
286
+ originalScore,
287
+ bestScore,
288
+ iterations,
289
+ reason,
290
+ diff: generateDiff(originalContent, bestContent),
291
+ };
292
+ } catch (err) {
293
+ try {
294
+ fs.writeFileSync(commandPath, bestContent, 'utf8');
295
+ } catch (_restoreErr) {
296
+ // Preserve the original failure if restoring also fails.
297
+ }
298
+ throw err;
299
+ }
300
+ }
301
+
302
+ // ---------------------------------------------------------------------------
303
+ // CLI entry point
304
+ // ---------------------------------------------------------------------------
305
+
306
+ /**
307
+ * Parse CLI arguments.
308
+ *
309
+ * @param {string[]} argv - process.argv.slice(2)
310
+ * @returns {{ commandPath: string, evalSetPath: string, maxIterations: number }}
311
+ */
312
+ function parseCliArgs(argv) {
313
+ let commandPath = null;
314
+ let evalSetPath = null;
315
+ let maxIterations = 3;
316
+
317
+ for (let i = 0; i < argv.length; i++) {
318
+ if (argv[i] === '--eval-set' && i + 1 < argv.length) {
319
+ evalSetPath = argv[i + 1];
320
+ i++;
321
+ } else if (argv[i] === '--max-iterations' && i + 1 < argv.length) {
322
+ maxIterations = Number(argv[i + 1]);
323
+ i++;
324
+ } else if (!argv[i].startsWith('--')) {
325
+ commandPath = argv[i];
326
+ }
327
+ }
328
+
329
+ if (!commandPath) {
330
+ throw new Error('Usage: improve-command <command-path> --eval-set <path> [--max-iterations <N>]');
331
+ }
332
+ if (!evalSetPath) {
333
+ throw new Error('Missing required --eval-set <path>');
334
+ }
335
+
336
+ return { commandPath, evalSetPath, maxIterations };
337
+ }
338
+
339
+ if (require.main === module) {
340
+ const args = parseCliArgs(process.argv.slice(2));
341
+ const { runEvalPipeline } = require('./run-command-eval');
342
+
343
+ runImprovementLoop(args.commandPath, args.evalSetPath, {
344
+ maxIterations: args.maxIterations,
345
+ _runEval: (evalPath) => runEvalPipeline(evalPath),
346
+ })
347
+ .then((result) => {
348
+ console.log('\n=== Improvement Summary ===');
349
+ console.log(`Original score: ${result.originalScore.toFixed(2)}`);
350
+ console.log(`Best score: ${result.bestScore.toFixed(2)}`);
351
+ console.log(`Iterations: ${result.iterations}`);
352
+ console.log(`Reason: ${result.reason}`);
353
+
354
+ if (result.diff) {
355
+ console.log('\n=== Diff (original -> best) ===');
356
+ console.log(result.diff);
357
+ } else {
358
+ console.log('\nNo changes made.');
359
+ }
360
+
361
+ console.log('\nDiff shown above. Review and decide whether to apply.');
362
+ })
363
+ .catch((err) => {
364
+ console.error(`Error: ${err.message}`);
365
+ process.exit(1);
366
+ });
367
+ }
368
+
369
+ module.exports = {
370
+ analyzeFailures,
371
+ buildRewritePrompt,
372
+ defaultRewriteCommand,
373
+ generateDiff,
374
+ runImprovementLoop,
375
+ };
@@ -0,0 +1,268 @@
1
+ /**
2
+ * Eval runner core — worktree isolation + command execution.
3
+ *
4
+ * Provides building blocks for the eval pipeline:
5
+ * - createEvalWorktree() — spin up an isolated worktree
6
+ * - destroyEvalWorktree() — tear it down (force, even if dirty)
7
+ * - resetWorktree() — reset between eval queries
8
+ * - executeCommand() — run a claude CLI command in a worktree
9
+ */
10
+
11
+ const path = require('path');
12
+ const { execSync } = require('node:child_process');
13
+
14
+ // ── active worktree tracking (cleanup on crash) ─────────────────────
15
+ // Tracks active eval worktrees so we can clean up on process exit/crash.
16
+ // Prevents orphaned eval-* branches when interrupted.
17
+ // Note: execSync is safe here — all paths are internally generated, never user input.
18
+ const activeEvalWorktrees = new Map(); // path -> branch
19
+
20
+ function cleanupActiveWorktrees() {
21
+ if (activeEvalWorktrees.size === 0) return;
22
+ let repoRoot;
23
+ try { repoRoot = getRepoRoot(); } catch (_err) { return; }
24
+ for (const [wtPath, branch] of activeEvalWorktrees) {
25
+ try {
26
+ execSync(`git worktree remove --force "${wtPath}"`, { cwd: repoRoot, stdio: 'pipe' });
27
+ } catch (_err) { /* already removed */ }
28
+ if (branch && branch.startsWith('eval-')) {
29
+ try {
30
+ execSync(`git branch -D "${branch}"`, { cwd: repoRoot, stdio: 'pipe' });
31
+ } catch (_err) { /* already deleted */ }
32
+ }
33
+ }
34
+ try { execSync('git worktree prune', { cwd: repoRoot, stdio: 'pipe' }); } catch (_err) { /* ignore */ }
35
+ activeEvalWorktrees.clear();
36
+ }
37
+
38
+ process.on('exit', cleanupActiveWorktrees);
39
+ process.on('SIGINT', () => {
40
+ const hadWork = activeEvalWorktrees.size > 0;
41
+ cleanupActiveWorktrees();
42
+ if (hadWork) process.exit(130);
43
+ });
44
+ process.on('SIGTERM', () => {
45
+ const hadWork = activeEvalWorktrees.size > 0;
46
+ cleanupActiveWorktrees();
47
+ if (hadWork) process.exit(143);
48
+ });
49
+
50
+ // ── helpers ──────────────────────────────────────────────────────────
51
+
52
+ /**
53
+ * Detect the repo root by walking up from cwd.
54
+ * Works from both the main repo and from within worktrees.
55
+ */
56
+ function getRepoRoot() {
57
+ const root = execSync('git rev-parse --show-toplevel', {
58
+ encoding: 'utf-8',
59
+ stdio: ['pipe', 'pipe', 'pipe'],
60
+ }).trim();
61
+ return root;
62
+ }
63
+
64
+ /**
65
+ * Get the .worktrees directory path for eval worktrees.
66
+ * Eval worktrees live under <repo-root>/.worktrees/
67
+ */
68
+ function getWorktreesDir() {
69
+ const root = getRepoRoot();
70
+ return path.join(root, '.worktrees');
71
+ }
72
+
73
+ // ── createEvalWorktree ───────────────────────────────────────────────
74
+
75
+ /**
76
+ * Create a git worktree with a unique name for eval isolation.
77
+ *
78
+ * @returns {Promise<{ path: string, branch: string }>}
79
+ */
80
+ async function createEvalWorktree() {
81
+ const timestamp = Date.now();
82
+ const pid = process.pid;
83
+ const name = `eval-${timestamp}-${pid}`;
84
+ const branch = `eval-${timestamp}-${pid}`;
85
+ const worktreesDir = getWorktreesDir();
86
+ const wtPath = path.join(worktreesDir, name);
87
+
88
+ // Create the worktree with a detached HEAD first, then create branch
89
+ execSync(`git worktree add -b "${branch}" "${wtPath}" HEAD`, {
90
+ cwd: getRepoRoot(),
91
+ encoding: 'utf-8',
92
+ stdio: ['pipe', 'pipe', 'pipe'],
93
+ });
94
+
95
+ activeEvalWorktrees.set(wtPath, branch);
96
+ return { path: wtPath, branch };
97
+ }
98
+
99
+ // ── destroyEvalWorktree ──────────────────────────────────────────────
100
+
101
+ /**
102
+ * Remove a worktree and its temporary branch.
103
+ * Succeeds even if the worktree is dirty.
104
+ *
105
+ * @param {string} worktreePath — absolute path to the worktree
106
+ * @returns {Promise<void>}
107
+ */
108
+ async function destroyEvalWorktree(worktreePath) {
109
+ const repoRoot = getRepoRoot();
110
+
111
+ // Query actual branch for this worktree (more reliable than inferring from dir name)
112
+ let branch;
113
+ try {
114
+ branch = execSync('git branch --show-current', {
115
+ cwd: worktreePath,
116
+ encoding: 'utf-8',
117
+ stdio: ['pipe', 'pipe', 'pipe'],
118
+ }).trim();
119
+ } catch (_err) {
120
+ // Worktree may be corrupted — fall back to directory name
121
+ branch = path.basename(worktreePath);
122
+ }
123
+
124
+ // Remove the worktree (--force handles dirty state)
125
+ execSync(`git worktree remove --force "${worktreePath}"`, {
126
+ cwd: repoRoot,
127
+ encoding: 'utf-8',
128
+ stdio: ['pipe', 'pipe', 'pipe'],
129
+ });
130
+
131
+ // Prune to clean up references
132
+ execSync('git worktree prune', {
133
+ cwd: repoRoot,
134
+ encoding: 'utf-8',
135
+ stdio: ['pipe', 'pipe', 'pipe'],
136
+ });
137
+
138
+ activeEvalWorktrees.delete(worktreePath);
139
+
140
+ // Delete the temporary branch (force in case it's not fully merged)
141
+ if (branch && branch.startsWith('eval-')) {
142
+ try {
143
+ execSync(`git branch -D "${branch}"`, {
144
+ cwd: repoRoot,
145
+ encoding: 'utf-8',
146
+ stdio: ['pipe', 'pipe', 'pipe'],
147
+ });
148
+ } catch (_err) {
149
+ // Branch may already be gone — ignore
150
+ }
151
+ }
152
+ }
153
+
154
+ // ── resetWorktree ────────────────────────────────────────────────────
155
+
156
+ /**
157
+ * Reset a worktree to a clean state (tracked files restored, untracked removed).
158
+ *
159
+ * @param {string} worktreePath — absolute path to the worktree
160
+ * @returns {Promise<void>}
161
+ */
162
+ async function resetWorktree(worktreePath) {
163
+ // Restore tracked files
164
+ execSync('git checkout -- .', {
165
+ cwd: worktreePath,
166
+ encoding: 'utf-8',
167
+ stdio: ['pipe', 'pipe', 'pipe'],
168
+ });
169
+
170
+ // Remove untracked files, directories, and ignored files (full reset between runs)
171
+ execSync('git clean -fdx', {
172
+ cwd: worktreePath,
173
+ encoding: 'utf-8',
174
+ stdio: ['pipe', 'pipe', 'pipe'],
175
+ });
176
+ }
177
+
178
+ // ── executeCommand ───────────────────────────────────────────────────
179
+
180
+ /**
181
+ * Execute a command in an eval worktree.
182
+ *
183
+ * In production, runs `claude -p "<prompt>" --output-format stream-json --verbose --no-session-persistence`.
184
+ * Accepts an optional `cmdOverride` array for testing (avoids invoking real LLM).
185
+ *
186
+ * @param {string} _command — label only (e.g., "/status")
187
+ * @param {string} prompt — the prompt to send
188
+ * @param {string} worktreePath — absolute path to the worktree (used as cwd)
189
+ * @param {number} [timeout=120000] — timeout in milliseconds
190
+ * @param {string[]} [cmdOverride] — optional command array for testing
191
+ * @returns {Promise<{ stdout: string, stderr: string, exitCode: number, timedOut: boolean }>}
192
+ */
193
+ async function executeCommand(_command, prompt, worktreePath, timeout = 120000, cmdOverride) {
194
+ // Build the command to run
195
+ const cmd = cmdOverride || [
196
+ 'claude',
197
+ '-p',
198
+ prompt,
199
+ '--output-format',
200
+ 'stream-json',
201
+ '--verbose',
202
+ '--no-session-persistence',
203
+ ];
204
+
205
+ // Build environment: inherit current env, strip CLAUDECODE, set FORGE_EVAL
206
+ const env = { ...process.env };
207
+ delete env.CLAUDECODE;
208
+ env.FORGE_EVAL = '1';
209
+
210
+ // Spawn with Bun.spawn (array form, no shell interpolation)
211
+ const proc = Bun.spawn(cmd, { // eslint-disable-line no-undef -- Bun global provided by runtime
212
+ cwd: worktreePath,
213
+ env,
214
+ stdout: 'pipe',
215
+ stderr: 'pipe',
216
+ });
217
+
218
+ let timedOut = false;
219
+ let timeoutId;
220
+
221
+ // Set up timeout
222
+ const timeoutPromise = new Promise((resolve) => {
223
+ timeoutId = setTimeout(() => {
224
+ timedOut = true;
225
+ proc.kill();
226
+ resolve();
227
+ }, timeout);
228
+ });
229
+
230
+ // Wait for process to exit (or timeout)
231
+ const exitPromise = proc.exited.then(() => {
232
+ clearTimeout(timeoutId);
233
+ });
234
+
235
+ await Promise.race([exitPromise, timeoutPromise]);
236
+
237
+ // Read stdout and stderr
238
+ let stdout = '';
239
+ let stderr = '';
240
+
241
+ try {
242
+ stdout = await new Response(proc.stdout).text();
243
+ } catch (_err) {
244
+ // stream may be closed on kill — ignore
245
+ }
246
+
247
+ try {
248
+ stderr = await new Response(proc.stderr).text();
249
+ } catch (_err) {
250
+ // stream may be closed on kill — ignore
251
+ }
252
+
253
+ const exitCode = timedOut ? (proc.exitCode ?? 1) : proc.exitCode;
254
+
255
+ return {
256
+ stdout,
257
+ stderr,
258
+ exitCode,
259
+ timedOut,
260
+ };
261
+ }
262
+
263
+ module.exports = {
264
+ createEvalWorktree,
265
+ destroyEvalWorktree,
266
+ resetWorktree,
267
+ executeCommand,
268
+ };